diff --git a/.claude/skills/review-pr/SKILL.md b/.claude/skills/review-pr/SKILL.md new file mode 100644 index 00000000000..5463e34d064 --- /dev/null +++ b/.claude/skills/review-pr/SKILL.md @@ -0,0 +1,42 @@ +--- +name: review-pr +description: Reviews a PR or diff with multi-angle finders and adversarial verification, then reports a findings table, a merge/no-merge recommendation, required followups, and offers to create a follow-up PR. Use when the user types /review-pr [PR# | branch | path]. +--- + +# PR Review (bifrost custom) + +## Scope +If a PR number is given: `gh pr view --json title,body,author,baseRefName,headRefName,state,additions,deletions,changedFiles` and `gh pr diff ` define the scope. Otherwise use `git diff @{upstream}...HEAD` (fall back to `git diff dev...HEAD`, then `git diff HEAD` for uncommitted work). + +The diff is the only review scope. When an angle needs surrounding code, Read files in this checkout if it matches the PR branch, otherwise fetch via `gh`. + +## Process +1. **Find** - launch independent finder agents in parallel (Agent tool), each returning up to 6 candidates `{file, line, summary, failure_scenario}`: + - Correctness angles: line-by-line diff scan (read the enclosing function of every hunk); removed-behavior audit (what invariant did deleted/replaced code enforce, and where is it re-established?); cross-file caller/callee trace (grep for changed symbols, check every call site). + - Cleanup angles: reuse (does new code re-implement an existing helper? name it); simplification (redundant state, copy-paste variation, dead code); efficiency (wasted work, repeated I/O, hot-path cost); altitude (is the fix at the right layer, or a bandaid on shared infrastructure?); conventions (check AGENTS.md rules - only flag when you can quote the exact rule and the exact violating line). + - Finders must pass through every candidate with a nameable failure scenario; do not self-censor half-believed ones. +2. **Verify** - dedup candidates that share a mechanism (keep the most concrete scenario), then run one verifier agent per candidate. Each returns exactly one of CONFIRMED (quote the triggering inputs and the wrong output), PLAUSIBLE (mechanism real, trigger uncertain - state what would confirm it), or REFUTED (quote the line that proves it wrong). Drop REFUTED. + +## Output format (required) +1. **2-3 sentence overview** of what the PR does and its overall quality. +2. **Findings table**: columns `# | Severity | Location | Finding | Verdict`, ranked most-severe first, max 8 rows. Location is a clickable `file:line`. If nothing survived verification, say so plainly. +3. **Merge recommendation**: exactly one of **Merge**, **Merge after nits**, or **Not yet - request changes**, with a 1-2 sentence reason tied to specific finding numbers. Only CONFIRMED correctness findings may block a merge; PLAUSIBLE and cleanup findings are nits or followups. +4. **Followups required**: numbered list, each tagged "in this PR (blocking)" or "follow-up PR", with concrete file:line targets and a one-line fix description. +5. **Follow-up PR offer**: if any follow-up items exist, offer to prepare them on a new branch. NEVER commit or push - leave changes staged for the user (standing user rule), unless the user explicitly authorizes committing and opening the PR with `gh pr create`. +6. Briefly list refuted candidates with the one-line reason, so the user knows what was checked and cleared. + +## Post the review to GitHub (after user approval) + +When reviewing a GitHub PR, after presenting the output above, offer to publish it on the PR. Only post after the user approves (posting is outward-facing). When approved: + +1. Build a single review payload as JSON in the scratchpad and submit it with `gh api repos/{owner}/{repo}/pulls//reviews --input .json` (get owner/repo via `gh repo view --json nameWithOwner`). +2. `event`: use `REQUEST_CHANGES` when the merge recommendation is "Not yet", `COMMENT` for "Merge after nits", `APPROVE` for "Merge". +3. `body`: the overview, findings table, merge recommendation, suggested followups (tagged blocking vs follow-up PR), and the refuted-candidates list. +4. `comments`: one inline comment per finding whose file is part of the PR diff, anchored with `path`, `line` (new-file line number, computed from the diff hunk headers), and `side: "RIGHT"`. Include the concrete failure scenario and a suggested fix as a code block. Findings in files NOT touched by the diff cannot be inline - keep them in the review body (GitHub rejects comments on unchanged files). +5. Report the resulting review URL back to the user. + +## House rules +- Never dismiss a finding as "out of scope" - address it or give a concrete design reason. +- No em dashes anywhere in the output. +- Diagrams, if any, are mermaid blocks, never ASCII art. +- Capability claims about providers must cite fetched upstream docs. diff --git a/.gitignore b/.gitignore index 63b8ebf7a0b..b14674855f4 100644 --- a/.gitignore +++ b/.gitignore @@ -187,3 +187,71 @@ tests/cmd/seedvks/seedvks # routing harness ledgers (local run journals, never committed) tests/e2e/api/routing/ledger-* + + +# Build artifacts +*/build/ +*.so +*.dll +*.dylib + +# Go build cache +*.exe +*.exe~ +*.test +*.out + +# Dependency directories +vendor/ + +reports/* +!reports/.keep + +# See https://help.github.com/articles/ignoring-files/ for more about ignoring files. + +# dependencies +/node_modules +/.pnp +.pnp.* +.yarn/* +!.yarn/patches +!.yarn/plugins +!.yarn/releases +!.yarn/versions + +# testing +/coverage + +# build output +ui/.next/ +ui/out/ + +# production +ui/build + +# misc +.DS_Store +*.pem + +# debug +npm-debug.log* +yarn-debug.log* +yarn-error.log* +.pnpm-debug.log* + +# env files (can opt-in for committing if needed) +.env* + +# vercel +.vercel + +# typescript +*.tsbuildinfo + +# auto-generated TanStack Router route tree +ui/app/routeTree.gen.ts + + +bifrost-migration-cli + +tests/e2e/clis/reports \ No newline at end of file diff --git a/Makefile b/Makefile index 6cc602891ac..2720763bb37 100644 --- a/Makefile +++ b/Makefile @@ -823,18 +823,19 @@ test-framework: install-gotestsum ## Run framework tests @$(EXPOSE_ENV); \ $(ECHO) "$(GREEN)Running framework tests...$(NC)"; \ mkdir -p $(TEST_REPORTS_DIR); \ + rm -f $(TEST_REPORTS_DIR)/.framework-failed; \ cd framework && find . -name "*.go" -path "*/tests/*" -o -name "*_test.go" | head -1 > /dev/null && \ for dir in $$(find . -name "*_test.go" -exec dirname {} \; | sort -u); do \ pkg_name=$$(echo $$dir | sed 's|^\./||' | sed 's|/|-|g'); \ $(ECHO) "Testing $$dir..."; \ - cd $$dir && gotestsum \ + ( cd $$dir && gotestsum \ --format=$(GOTESTSUM_FORMAT) \ - --junitfile=../../$(TEST_REPORTS_DIR)/framework-$$pkg_name.xml \ - -- -v ./... && cd - > /dev/null; \ + --junitfile=$(CURDIR)/$(TEST_REPORTS_DIR)/framework-$$pkg_name.xml \ + -- -v ./... ) || touch $(CURDIR)/$(TEST_REPORTS_DIR)/.framework-failed; \ if [ -z "$$CI" ] && [ -z "$$GITHUB_ACTIONS" ] && [ -z "$$GITLAB_CI" ] && [ -z "$$CIRCLECI" ] && [ -z "$$JENKINS_HOME" ]; then \ if which junit-viewer > /dev/null 2>&1; then \ $(ECHO) "$(YELLOW)Generating HTML report for $$pkg_name...$(NC)"; \ - junit-viewer --results=../$(TEST_REPORTS_DIR)/framework-$$pkg_name.xml --save=../$(TEST_REPORTS_DIR)/framework-$$pkg_name.html 2>/dev/null || true; \ + junit-viewer --results=$(CURDIR)/$(TEST_REPORTS_DIR)/framework-$$pkg_name.xml --save=$(CURDIR)/$(TEST_REPORTS_DIR)/framework-$$pkg_name.html 2>/dev/null || true; \ fi; \ fi; \ done || $(ECHO) "No framework tests found" @@ -848,6 +849,10 @@ test-framework: install-gotestsum ## Run framework tests SUMMARY_LABEL="Framework" \ SUMMARY_STRIP="framework-" \ SUMMARY_FILES="$(TEST_REPORTS_DIR)/framework-*.xml" + @if [ -f $(TEST_REPORTS_DIR)/.framework-failed ]; then \ + rm -f $(TEST_REPORTS_DIR)/.framework-failed; \ + exit 1; \ + fi # Internal: render a table of test reports + a final pass/fail scenario. # Usage: $(MAKE) print-test-summary SUMMARY_LABEL="Framework" SUMMARY_STRIP="framework-" SUMMARY_FILES="" diff --git a/core/bifrost.go b/core/bifrost.go index c733816c674..5d512f9c8f7 100644 --- a/core/bifrost.go +++ b/core/bifrost.go @@ -24,8 +24,10 @@ import ( "github.com/maximhq/bifrost/core/providers/anthropic" "github.com/maximhq/bifrost/core/providers/azure" "github.com/maximhq/bifrost/core/providers/bedrock" + "github.com/maximhq/bifrost/core/providers/bedrockmantle" "github.com/maximhq/bifrost/core/providers/cerebras" "github.com/maximhq/bifrost/core/providers/cohere" + "github.com/maximhq/bifrost/core/providers/deepseek" "github.com/maximhq/bifrost/core/providers/elevenlabs" "github.com/maximhq/bifrost/core/providers/fireworks" "github.com/maximhq/bifrost/core/providers/gemini" @@ -40,8 +42,8 @@ import ( "github.com/maximhq/bifrost/core/providers/parasail" "github.com/maximhq/bifrost/core/providers/perplexity" "github.com/maximhq/bifrost/core/providers/replicate" - "github.com/maximhq/bifrost/core/providers/runway" "github.com/maximhq/bifrost/core/providers/runware" + "github.com/maximhq/bifrost/core/providers/runway" "github.com/maximhq/bifrost/core/providers/sgl" providerUtils "github.com/maximhq/bifrost/core/providers/utils" "github.com/maximhq/bifrost/core/providers/vertex" @@ -1009,6 +1011,203 @@ func (bifrost *Bifrost) CompactionRequest(ctx *schemas.BifrostContext, req *sche return response.CompactionResponse, nil } +// ResponsesRetrieveRequest retrieves a stored response by ID (OpenAI GET /v1/responses/{id}). +func (bifrost *Bifrost) ResponsesRetrieveRequest(ctx *schemas.BifrostContext, req *schemas.BifrostResponsesRetrieveRequest) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) { + if req == nil { + return nil, &schemas.BifrostError{ + IsBifrostError: false, + Error: &schemas.ErrorField{ + Message: "responses retrieve request is nil", + }, + ExtraFields: schemas.BifrostErrorExtraFields{ + RequestType: schemas.ResponsesRetrieveRequest, + }, + } + } + if ctx == nil { + ctx = bifrost.ctx + } + if req.Provider == "" { + return nil, &schemas.BifrostError{ + IsBifrostError: false, + Error: &schemas.ErrorField{ + Message: "provider is required for responses retrieve request", + }, + ExtraFields: schemas.BifrostErrorExtraFields{ + RequestType: schemas.ResponsesRetrieveRequest, + Provider: req.Provider, + }, + } + } + if req.ResponseID == "" { + return nil, &schemas.BifrostError{ + IsBifrostError: false, + Error: &schemas.ErrorField{ + Message: "response_id is required for responses retrieve request", + }, + ExtraFields: schemas.BifrostErrorExtraFields{ + RequestType: schemas.ResponsesRetrieveRequest, + Provider: req.Provider, + }, + } + } + bifrostReq := bifrost.getBifrostRequest() + bifrostReq.RequestType = schemas.ResponsesRetrieveRequest + bifrostReq.ResponsesRetrieveRequest = req + response, err := bifrost.handleRequest(ctx, bifrostReq) + if err != nil { + return nil, err + } + return response.ResponsesResponse, nil +} + +// ResponsesDeleteRequest deletes a stored response (OpenAI DELETE /v1/responses/{id}). +func (bifrost *Bifrost) ResponsesDeleteRequest(ctx *schemas.BifrostContext, req *schemas.BifrostResponsesDeleteRequest) (*schemas.BifrostResponsesDeleteResponse, *schemas.BifrostError) { + if req == nil { + return nil, &schemas.BifrostError{ + IsBifrostError: false, + Error: &schemas.ErrorField{ + Message: "responses delete request is nil", + }, + ExtraFields: schemas.BifrostErrorExtraFields{ + RequestType: schemas.ResponsesDeleteRequest, + }, + } + } + if ctx == nil { + ctx = bifrost.ctx + } + if req.Provider == "" { + return nil, &schemas.BifrostError{ + IsBifrostError: false, + Error: &schemas.ErrorField{ + Message: "provider is required for responses delete request", + }, + ExtraFields: schemas.BifrostErrorExtraFields{ + RequestType: schemas.ResponsesDeleteRequest, + }, + } + } + if req.ResponseID == "" { + return nil, &schemas.BifrostError{ + IsBifrostError: false, + Error: &schemas.ErrorField{ + Message: "response_id is required for responses delete request", + }, + ExtraFields: schemas.BifrostErrorExtraFields{ + RequestType: schemas.ResponsesDeleteRequest, + Provider: req.Provider, + }, + } + } + bifrostReq := bifrost.getBifrostRequest() + bifrostReq.RequestType = schemas.ResponsesDeleteRequest + bifrostReq.ResponsesDeleteRequest = req + response, err := bifrost.handleRequest(ctx, bifrostReq) + if err != nil { + return nil, err + } + return response.ResponsesDeleteResponse, nil +} + +// ResponsesCancelRequest cancels an in-flight stored response (OpenAI POST /v1/responses/{id}/cancel). +func (bifrost *Bifrost) ResponsesCancelRequest(ctx *schemas.BifrostContext, req *schemas.BifrostResponsesCancelRequest) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) { + if req == nil { + return nil, &schemas.BifrostError{ + IsBifrostError: false, + Error: &schemas.ErrorField{ + Message: "responses cancel request is nil", + }, + ExtraFields: schemas.BifrostErrorExtraFields{ + RequestType: schemas.ResponsesCancelRequest, + }, + } + } + if ctx == nil { + ctx = bifrost.ctx + } + if req.Provider == "" { + return nil, &schemas.BifrostError{ + IsBifrostError: false, + Error: &schemas.ErrorField{ + Message: "provider is required for responses cancel request", + }, + ExtraFields: schemas.BifrostErrorExtraFields{ + RequestType: schemas.ResponsesCancelRequest, + }, + } + } + if req.ResponseID == "" { + return nil, &schemas.BifrostError{ + IsBifrostError: false, + Error: &schemas.ErrorField{ + Message: "response_id is required for responses cancel request", + }, + ExtraFields: schemas.BifrostErrorExtraFields{ + RequestType: schemas.ResponsesCancelRequest, + Provider: req.Provider, + }, + } + } + bifrostReq := bifrost.getBifrostRequest() + bifrostReq.RequestType = schemas.ResponsesCancelRequest + bifrostReq.ResponsesCancelRequest = req + response, err := bifrost.handleRequest(ctx, bifrostReq) + if err != nil { + return nil, err + } + return response.ResponsesResponse, nil +} + +// ResponsesInputItemsRequest lists input items for a stored response (OpenAI GET /v1/responses/{id}/input_items). +func (bifrost *Bifrost) ResponsesInputItemsRequest(ctx *schemas.BifrostContext, req *schemas.BifrostResponsesInputItemsRequest) (*schemas.BifrostResponsesInputItemsResponse, *schemas.BifrostError) { + if req == nil { + return nil, &schemas.BifrostError{ + IsBifrostError: false, + Error: &schemas.ErrorField{ + Message: "responses input items request is nil", + }, + ExtraFields: schemas.BifrostErrorExtraFields{ + RequestType: schemas.ResponsesInputItemsRequest, + }, + } + } + if ctx == nil { + ctx = bifrost.ctx + } + if req.Provider == "" { + return nil, &schemas.BifrostError{ + IsBifrostError: false, + Error: &schemas.ErrorField{ + Message: "provider is required for responses input items request", + }, + ExtraFields: schemas.BifrostErrorExtraFields{ + RequestType: schemas.ResponsesInputItemsRequest, + }, + } + } + if req.ResponseID == "" { + return nil, &schemas.BifrostError{ + IsBifrostError: false, + Error: &schemas.ErrorField{ + Message: "response_id is required for responses input items request", + }, + ExtraFields: schemas.BifrostErrorExtraFields{ + RequestType: schemas.ResponsesInputItemsRequest, + Provider: req.Provider, + }, + } + } + bifrostReq := bifrost.getBifrostRequest() + bifrostReq.RequestType = schemas.ResponsesInputItemsRequest + bifrostReq.ResponsesInputItemsRequest = req + response, err := bifrost.handleRequest(ctx, bifrostReq) + if err != nil { + return nil, err + } + return response.ResponsesInputItemsResponse, nil +} + // EmbeddingRequest sends an embedding request to the specified provider. func (bifrost *Bifrost) EmbeddingRequest(ctx *schemas.BifrostContext, req *schemas.BifrostEmbeddingRequest) (*schemas.BifrostEmbeddingResponse, *schemas.BifrostError) { if req == nil { @@ -2577,6 +2776,34 @@ func (bifrost *Bifrost) PassthroughStream( return bifrost.handleStreamRequest(ctx, bifrostReq) } +// ensureMCPRawStorageContext sets BifrostContextKeyShouldStoreRawInLogs for standalone MCP +// tool executions so PostMCPHook consumers (e.g. the logging plugin) see an explicit value. +// In-pipeline tool calls already carry the key from the LLM request path (see the effective +// raw-storage computation in requestWorker), so an existing value is never overwritten. There is +// no provider config on the standalone path, so the default is false and only the per-request +// override is honored, mirroring the per-request half of the LLM pipeline's logic. +// +// Callers must only pass request-scoped contexts, never the shared instance context +// (bifrost.ctx): SetValue mutates the receiver's value map, so writing here would stamp the +// flag onto state shared by unrelated calls. Nil-ctx standalone executions therefore skip this +// entirely — consumers treat a missing key as false, and the per-request override keys can +// never be present on the instance context, so the outcome is identical. +func ensureMCPRawStorageContext(ctx *schemas.BifrostContext) { + if ctx == nil { + return + } + if _, ok := ctx.Value(schemas.BifrostContextKeyShouldStoreRawInLogs).(bool); ok { + return + } + effectiveStore := false + if allowStorageOverride, _ := ctx.Value(schemas.BifrostContextKeyAllowPerRequestStorageOverride).(bool); allowStorageOverride { + if override, ok := ctx.Value(schemas.BifrostContextKeyStoreRawRequestResponse).(bool); ok { + effectiveStore = override + } + } + ctx.SetValue(schemas.BifrostContextKeyShouldStoreRawInLogs, effectiveStore) +} + // ExecuteChatMCPTool executes an MCP tool call and returns the result as a chat message. // This is the main public API for manual MCP tool execution in Chat format. All the // real work — request pooling, plugin gate (PreMCPHook / PostMCPHook), short-circuit @@ -2584,6 +2811,8 @@ func (bifrost *Bifrost) PassthroughStream( func (bifrost *Bifrost) ExecuteChatMCPTool(ctx *schemas.BifrostContext, toolCall *schemas.ChatAssistantMessageToolCall) (*schemas.ChatMessage, *schemas.BifrostError) { if ctx == nil { ctx = bifrost.ctx + } else { + ensureMCPRawStorageContext(ctx) } if bifrost.MCPManager == nil { return nil, &schemas.BifrostError{ @@ -2600,6 +2829,8 @@ func (bifrost *Bifrost) ExecuteChatMCPTool(ctx *schemas.BifrostContext, toolCall func (bifrost *Bifrost) ExecuteResponsesMCPTool(ctx *schemas.BifrostContext, toolCall *schemas.ResponsesToolMessage) (*schemas.ResponsesMessage, *schemas.BifrostError) { if ctx == nil { ctx = bifrost.ctx + } else { + ensureMCPRawStorageContext(ctx) } if bifrost.MCPManager == nil { return nil, &schemas.BifrostError{ @@ -4010,6 +4241,8 @@ func (bifrost *Bifrost) createBaseProvider(providerKey schemas.ModelProvider, co return anthropic.NewAnthropicProvider(config, bifrost.logger), nil case schemas.Bedrock: return bedrock.NewBedrockProvider(config, bifrost.logger) + case schemas.BedrockMantle: + return bedrockmantle.NewBedrockMantleProvider(config, bifrost.logger) case schemas.Cohere: return cohere.NewCohereProvider(config, bifrost.logger) case schemas.Azure: @@ -4034,6 +4267,8 @@ func (bifrost *Bifrost) createBaseProvider(providerKey schemas.ModelProvider, co return perplexity.NewPerplexityProvider(config, bifrost.logger) case schemas.Cerebras: return cerebras.NewCerebrasProvider(config, bifrost.logger) + case schemas.DeepSeek: + return deepseek.NewDeepSeekProvider(config, bifrost.logger) case schemas.Gemini: return gemini.NewGeminiProvider(config, bifrost.logger), nil case schemas.OpenRouter: @@ -5039,12 +5274,7 @@ func (bifrost *Bifrost) tryRequest(ctx *schemas.BifrostContext, req *schemas.Bif if reroutedPq == nil { bifrost.releaseChannelMessage(msg) bifrostErr := newBifrostErrorFromMsg("provider is shutting down") - bifrostErr.ExtraFields = schemas.BifrostErrorExtraFields{ - RequestType: req.RequestType, - Provider: provider, - OriginalModelRequested: model, - ResolvedModelUsed: model, - } + bifrostErr.PopulateExtraFields(req.RequestType, provider, model, model) return nil, bifrostErr } pq = reroutedPq @@ -5378,11 +5608,7 @@ func (bifrost *Bifrost) tryStreamRequest(ctx *schemas.BifrostContext, req *schem if reroutedPq == nil { bifrost.releaseChannelMessage(msg) bifrostErr := newBifrostErrorFromMsg("provider is shutting down") - bifrostErr.ExtraFields = schemas.BifrostErrorExtraFields{ - RequestType: req.RequestType, - Provider: provider, - OriginalModelRequested: model, - } + bifrostErr.PopulateExtraFields(req.RequestType, provider, model, model) return nil, bifrostErr } pq = reroutedPq @@ -5789,6 +6015,24 @@ func executeRequestWithRetries[T any]( if businessUnitName, ok := ctx.Value(schemas.BifrostContextKeyGovernanceBusinessUnitName).(string); ok && businessUnitName != "" { tracer.SetAttribute(handle, schemas.AttrBifrostBusinessUnitName, businessUnitName) } + if teamIDs, ok := ctx.Value(schemas.BifrostContextKeyGovernanceTeamIDs).([]string); ok && len(teamIDs) > 0 { + tracer.SetAttribute(handle, schemas.AttrBifrostTeamIDs, teamIDs) + } + if teamNames, ok := ctx.Value(schemas.BifrostContextKeyGovernanceTeamNames).([]string); ok && len(teamNames) > 0 { + tracer.SetAttribute(handle, schemas.AttrBifrostTeamNames, teamNames) + } + if customerIDs, ok := ctx.Value(schemas.BifrostContextKeyGovernanceCustomerIDs).([]string); ok && len(customerIDs) > 0 { + tracer.SetAttribute(handle, schemas.AttrBifrostCustomerIDs, customerIDs) + } + if customerNames, ok := ctx.Value(schemas.BifrostContextKeyGovernanceCustomerNames).([]string); ok && len(customerNames) > 0 { + tracer.SetAttribute(handle, schemas.AttrBifrostCustomerNames, customerNames) + } + if businessUnitIDs, ok := ctx.Value(schemas.BifrostContextKeyGovernanceBusinessUnitIDs).([]string); ok && len(businessUnitIDs) > 0 { + tracer.SetAttribute(handle, schemas.AttrBifrostBusinessUnitIDs, businessUnitIDs) + } + if businessUnitNames, ok := ctx.Value(schemas.BifrostContextKeyGovernanceBusinessUnitNames).([]string); ok && len(businessUnitNames) > 0 { + tracer.SetAttribute(handle, schemas.AttrBifrostBusinessUnitNames, businessUnitNames) + } if userID, ok := ctx.Value(schemas.BifrostContextKeyUserID).(string); ok && userID != "" { tracer.SetAttribute(handle, schemas.AttrBifrostUserID, userID) } @@ -5853,6 +6097,12 @@ func executeRequestWithRetries[T any]( checkedStream, drainDone, firstChunkErr := providerUtils.CheckFirstStreamChunkForError(ctx, streamChan) if firstChunkErr != nil { <-drainDone + // The dead stream's teardown (ReleaseStreamingResponse) claimed the + // connection_closed flag on the shared context. That claim is scoped + // to the response it released; clear it so the retry or fallback + // attempt that follows doesn't see its own fresh stream as already + // closed and fail every read with ErrStreamClosed. + ctx.ClearValue(schemas.BifrostContextKeyConnectionClosed) bifrostError = firstChunkErr } else { result = any(checkedStream).(T) @@ -6036,7 +6286,10 @@ func clearAnthropicPassthroughForNonNativeProvider(ctx *schemas.BifrostContext, if integrationType, _ := ctx.Value(schemas.BifrostContextKeyIntegrationType).(string); integrationType != "anthropic" { return } - if baseProvider == schemas.Anthropic || baseProvider == schemas.Vertex || baseProvider == schemas.Azure { + if baseProvider == schemas.Anthropic || + baseProvider == schemas.Vertex || + baseProvider == schemas.Azure || + baseProvider == schemas.BedrockMantle { return } ctx.SetValue(schemas.BifrostContextKeyUseRawRequestBody, false) @@ -6620,6 +6873,46 @@ func (bifrost *Bifrost) handleProviderRequest(provider schemas.Provider, config return nil, bifrostError } response.CountTokensResponse = countTokensResponse + case schemas.ResponsesRetrieveRequest: + lifecycle, ok := provider.(schemas.ResponsesLifecycleProvider) + if !ok { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ResponsesRetrieveRequest, provider.GetProviderKey()) + } + retrieveResp, bifrostError := lifecycle.ResponsesRetrieve(req.Context, key, req.BifrostRequest.ResponsesRetrieveRequest) + if bifrostError != nil { + return nil, bifrostError + } + response.ResponsesResponse = retrieveResp + case schemas.ResponsesDeleteRequest: + lifecycle, ok := provider.(schemas.ResponsesLifecycleProvider) + if !ok { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ResponsesDeleteRequest, provider.GetProviderKey()) + } + deleteResp, bifrostError := lifecycle.ResponsesDelete(req.Context, key, req.BifrostRequest.ResponsesDeleteRequest) + if bifrostError != nil { + return nil, bifrostError + } + response.ResponsesDeleteResponse = deleteResp + case schemas.ResponsesCancelRequest: + lifecycle, ok := provider.(schemas.ResponsesLifecycleProvider) + if !ok { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ResponsesCancelRequest, provider.GetProviderKey()) + } + cancelResp, bifrostError := lifecycle.ResponsesCancel(req.Context, key, req.BifrostRequest.ResponsesCancelRequest) + if bifrostError != nil { + return nil, bifrostError + } + response.ResponsesResponse = cancelResp + case schemas.ResponsesInputItemsRequest: + lifecycle, ok := provider.(schemas.ResponsesLifecycleProvider) + if !ok { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ResponsesInputItemsRequest, provider.GetProviderKey()) + } + itemsResp, bifrostError := lifecycle.ResponsesInputItems(req.Context, key, req.BifrostRequest.ResponsesInputItemsRequest) + if bifrostError != nil { + return nil, bifrostError + } + response.ResponsesInputItemsResponse = itemsResp case schemas.CompactionRequest: compactionResponse, bifrostError := provider.Compaction(req.Context, key, req.BifrostRequest.CompactionRequest) if bifrostError != nil { @@ -7688,6 +7981,10 @@ func resetBifrostRequest(req *schemas.BifrostRequest) { req.TextCompletionRequest = nil req.ChatRequest = nil req.ResponsesRequest = nil + req.ResponsesRetrieveRequest = nil + req.ResponsesDeleteRequest = nil + req.ResponsesCancelRequest = nil + req.ResponsesInputItemsRequest = nil req.CountTokensRequest = nil req.CompactionRequest = nil req.EmbeddingRequest = nil @@ -7931,7 +8228,7 @@ func (bifrost *Bifrost) selectKeyFromProviderForModelWithPool(ctx *schemas.Bifro // Skip model check conditions // We can improve these conditions in the future - skipModelCheck := (model == "" && (isFileRequestType(requestType) || isBatchRequestType(requestType) || isContainerRequestType(requestType) || isCachedContentRequestType(requestType) || isModellessVideoRequestType(requestType) || isPassthroughRequestType(requestType))) || requestType == schemas.ListModelsRequest + skipModelCheck := (model == "" && (isFileRequestType(requestType) || isBatchRequestType(requestType) || isContainerRequestType(requestType) || isCachedContentRequestType(requestType) || isModellessVideoRequestType(requestType) || isPassthroughRequestType(requestType))) || requestType == schemas.ListModelsRequest || isResponsesLifecycleRequestType(requestType) if skipModelCheck { // When skipping model check: just verify keys are enabled and have values for _, key := range keys { diff --git a/core/changelog.md b/core/changelog.md index e69de29bb2d..f7f09ce7a1e 100644 --- a/core/changelog.md +++ b/core/changelog.md @@ -0,0 +1,46 @@ +- feat: added DeepSeek as a first-class provider (#4852) +- feat: added `bedrock_mantle` as a first-class provider with native-Anthropic and OpenAI-compatible routing and SigV4 key config (#4736, #4737) +- feat: added JWT Bearer authentication path for `/mcp` with session validation, a `virtualKeysByID` secondary index, and cached signing key and VK lookups (#4508, #4783) +- feat: added per-MCP-server tool execution timeout (#4472, closes #4446) (thanks [@Purvi09](https://github.com/Purvi09)!) +- feat: added missing OpenAI Responses lifecycle methods (#3125, closes #3121) (thanks [@17jmumford](https://github.com/17jmumford)!) +- feat: added IPv6 support (#4895) +- feat: added ClickHouse support for the log store (#4748) +- feat: extended Bedrock vendor-prefix pricing fallback to OpenAI, Google, and xAI models and folded `bedrock_mantle` onto `bedrock` lookups (#4924) +- feat: added `is_deprecated` to model pricing and catalog responses and mark deprecated models instead of filtering them (#4779, #4792, #4936) +- feat: added latency info on errors (#4867, #4876) +- feat: added multiple teams, customers, and business units to connectors (#4875) +- feat: virtual key values use `schemas.SecretVar` to support the env store (#4817) +- feat: `chunking_strategy` passes through as an extra param for OpenAI models (#4741, closes #4720) +- feat: simplified Responses lifecycle permissions to explicit per-verb flags (#4880) +- fix: round-trip Anthropic `redacted_thinking` blocks on chat completions so tool-use turns with redacted reasoning can be replayed (#4943, closes #4942) (thanks [@fus3r](https://github.com/fus3r)!) +- fix: emit `contentBlockStop` events on the Bedrock ConverseStream egress so consumers that assemble messages on block boundaries get complete content (#4923, closes #4262) (thanks [@fus3r](https://github.com/fus3r)!) +- fix: clear the per-attempt stream close claim so streaming retries and fallbacks are not dead on arrival after an SSE-embedded provider error (#4911, closes #4788) (thanks [@fus3r](https://github.com/fus3r)!) +- fix: emit reads-only `cached_tokens` in usage per the OpenAI spec so cache writes are not priced as cache reads (#4906, closes #4816) (thanks [@fus3r](https://github.com/fus3r)!) +- fix: forward file IDs and content type on the Anthropic files integration (#4956) +- fix: preserve Anthropic file ID document sources (#4832) (thanks [@mmacvicar](https://github.com/mmacvicar)!) +- fix: preserve Gemini file upload MIME types for GenAI file URI completions (#4833) (thanks [@mmacvicar](https://github.com/mmacvicar)!) +- fix: propagate `max_tokens` from the OpenAI integration (#4966) +- fix: error type setting in all integrations for Bedrock (#4958) +- fix: guard setting tool call config in Gemini (#4959) +- fix: Gemini 2.5-pro thinking budget value (#4947) +- fix: Gemini OpenAI-through signature compatibility (#4810) +- fix: surface Gemini batch inline responses from the response field, not dest (#4904, closes #3951) (thanks [@nnNyx](https://github.com/nnNyx)!) +- fix: Gemini video reference fields map to instances (thanks [@vojthor](https://github.com/vojthor)!) +- fix: web fetch fixes (#4945) +- fix: set `SecretTypePlainText` for plain-text JSON and non-prefixed secret values (#4946) and check whether virtual key values are secrets (#4927) +- fix: idle timeout wiring in the Vertex path and recover from idle-timeout timer-goroutine panic (#4937) +- fix: set content type header consistently for Responses API requests (#4935) +- fix: deterministic MCP tool ordering for prompt cache stability (#4932, closes #2347) +- fix: consistent content_block indices for server tools on Claude Code passthrough streaming (#4890) (thanks [@surki](https://github.com/surki)!) +- fix: empty tool call result insertion failures (#4925) +- fix: sanitize error details on the log update path and set the raw-storage log flag on standalone MCP tool executions (#4913) +- fix: complete deferred LLM span on streaming goroutine exit (#4885) +- fix: billing on failed Responses stream requests for Anthropic and Bedrock (#4842) +- fix: cost for image generation and image edit streaming (#4802, closes #4777) +- fix: Perplexity Responses API compatibility (#4813) +- fix: signal Bedrock max_output_tokens truncation on the Responses API (#4680, closes #4679) (thanks [@jeremym-tanium](https://github.com/jeremym-tanium)!) +- fix: preserve codex `tool_search_call` and `tool_search_output` input items and accept object-valued tool-call arguments on the Responses API streaming path (#4121) (thanks [@raghu-nandan-bs](https://github.com/raghu-nandan-bs)!) +- fix: MCP reconnect failure on startup (#4316, closes #4314) (thanks [@HackToHell](https://github.com/HackToHell)!) +- fix: pass through `gs://` image URLs on Vertex Gemini (#4568, closes #4402) (thanks [@G-XD](https://github.com/G-XD)!) +- fix: skip model check for Responses lifecycle APIs (#4920) +- chore: refactored Anthropic request building into `BuildAnthropicChatRequestBody`, shared `completeRequest` across Anthropic, Azure, and Bedrock, lazy `BodySigner` SigV4 signing, and `BearerAuthHeader` helper (#3309, #4394, #4425, #4735) diff --git a/core/internal/llmtests/account.go b/core/internal/llmtests/account.go index 4846bacf72d..e0a10361418 100644 --- a/core/internal/llmtests/account.go +++ b/core/internal/llmtests/account.go @@ -99,6 +99,7 @@ type TestScenarios struct { FastMode bool // Fast mode for Opus 4.6 (beta: research preview) EagerInputStreaming bool // Fine-grained tool input streaming (Anthropic fine-grained-tool-streaming-2025-05-14) ServerToolsViaOpenAIEndpoint bool // Anthropic server-tool shapes in tools[] via /v1/chat/completions (web_search / web_fetch / code_execution) + ResponsesLifecycle bool // OpenAI GET/DELETE responses + input_items lifecycle (stored responses) } // ComprehensiveTestConfig extends TestConfig with additional scenarios @@ -162,6 +163,7 @@ func (account *ComprehensiveTestAccount) GetConfiguredProviders() ([]schemas.Mod schemas.OpenAI, schemas.Anthropic, schemas.Bedrock, + schemas.BedrockMantle, schemas.Cohere, schemas.Azure, schemas.Vertex, @@ -173,6 +175,7 @@ func (account *ComprehensiveTestAccount) GetConfiguredProviders() ([]schemas.Mod schemas.Elevenlabs, schemas.Perplexity, schemas.Cerebras, + schemas.DeepSeek, schemas.Gemini, schemas.OpenRouter, schemas.HuggingFace, @@ -290,6 +293,28 @@ func (account *ComprehensiveTestAccount) GetKeysForProvider(ctx context.Context, }, }, }, nil + case schemas.BedrockMantle: + // A single Bedrock Mantle endpoint serves the whole catalog (see /v1/models), so one key + // serves all models ("*"). The native-Anthropic surface uses "anthropic.{model}" ids (no + // cross-region prefix or version suffix); the OpenAI-compatible surface uses + // "openai.{model}" / "google.{model}". + return []schemas.Key{ + { + Models: []string{"*"}, + Weight: 1.0, + BedrockMantleKeyConfig: &schemas.BedrockMantleKeyConfig{ + AccessKey: *schemas.NewSecretVar("env.AWS_ACCESS_KEY_ID"), + SecretKey: *schemas.NewSecretVar("env.AWS_SECRET_ACCESS_KEY"), + SessionToken: schemas.NewSecretVar("env.AWS_SESSION_TOKEN"), + Region: schemas.NewSecretVar(getEnvWithDefault("AWS_REGION", "us-east-1")), + }, + // Mantle does not support batch/file ops, but the key must pass the batch-key + // filter so those requests reach the provider's unsupported-operation stub + // (the BatchUnsupported/FileUnsupported harness checks), as other non-batch + // providers (e.g. Cohere) do. + UseForBatchAPI: bifrost.Ptr(true), + }, + }, nil case schemas.Cohere: return []schemas.Key{ { @@ -434,6 +459,15 @@ func (account *ComprehensiveTestAccount) GetKeysForProvider(ctx context.Context, UseForBatchAPI: bifrost.Ptr(true), }, }, nil + case schemas.DeepSeek: + return []schemas.Key{ + { + Value: *schemas.NewSecretVar("env.DEEPSEEK_API_KEY"), + Models: []string{"*"}, + Weight: 1.0, + UseForBatchAPI: bifrost.Ptr(true), + }, + }, nil case schemas.Gemini: return []schemas.Key{ { @@ -604,6 +638,19 @@ func (account *ComprehensiveTestAccount) GetConfigForProvider(providerKey schema BufferSize: 10, }, }, nil + case schemas.BedrockMantle: + return &schemas.ProviderConfig{ + NetworkConfig: schemas.NetworkConfig{ + DefaultRequestTimeoutInSeconds: 120, + MaxRetries: 10, // AWS services can have occasional issues + RetryBackoffInitial: 5 * time.Second, + RetryBackoffMax: 40 * time.Second, + }, + ConcurrencyAndBufferSize: schemas.ConcurrencyAndBufferSize{ + Concurrency: Concurrency, + BufferSize: 10, + }, + }, nil case schemas.Cohere: return &schemas.ProviderConfig{ NetworkConfig: schemas.NetworkConfig{ @@ -749,6 +796,19 @@ func (account *ComprehensiveTestAccount) GetConfigForProvider(providerKey schema BufferSize: 10, }, }, nil + case schemas.DeepSeek: + return &schemas.ProviderConfig{ + NetworkConfig: schemas.NetworkConfig{ + DefaultRequestTimeoutInSeconds: 120, + MaxRetries: 10, + RetryBackoffInitial: 5 * time.Second, + RetryBackoffMax: 3 * time.Minute, + }, + ConcurrencyAndBufferSize: schemas.ConcurrencyAndBufferSize{ + Concurrency: Concurrency, + BufferSize: 10, + }, + }, nil case schemas.VLLM: return &schemas.ProviderConfig{ NetworkConfig: schemas.NetworkConfig{ @@ -948,6 +1008,7 @@ var AllProviderConfigs = []ComprehensiveTestConfig{ ContainerFileRetrieve: true, // OpenAI supports container file API ContainerFileContent: true, // OpenAI supports container file API ContainerFileDelete: true, // OpenAI supports container file API + ResponsesLifecycle: true, // OpenAI stored response retrieve/delete/input_items }, Fallbacks: []schemas.Fallback{ {Provider: schemas.Anthropic, Model: "claude-3-7-sonnet-20250219"}, diff --git a/core/internal/llmtests/responses_lifecycle.go b/core/internal/llmtests/responses_lifecycle.go new file mode 100644 index 00000000000..455340aa788 --- /dev/null +++ b/core/internal/llmtests/responses_lifecycle.go @@ -0,0 +1,92 @@ +package llmtests + +import ( + "context" + "testing" + + bifrost "github.com/maximhq/bifrost/core" + "github.com/maximhq/bifrost/core/schemas" +) + +// RunResponsesLifecycleTest exercises OpenAI Responses API lifecycle: create with store, +// retrieve, list input_items, delete. Cancel is only meaningful for background responses and is omitted. +func RunResponsesLifecycleTest(t *testing.T, client *bifrost.Bifrost, ctx context.Context, testConfig ComprehensiveTestConfig) { + if !testConfig.Scenarios.ResponsesLifecycle { + return + } + if testConfig.Provider != schemas.OpenAI { + t.Skip("responses lifecycle is only run for OpenAI provider") + } + + model := testConfig.ChatModel + if model == "" { + model = "gpt-4o-mini" + } + + bfCtx := schemas.NewBifrostContext(ctx, schemas.NoDeadline) + store := true + createReq := &schemas.BifrostResponsesRequest{ + Provider: testConfig.Provider, + Model: model, + Input: []schemas.ResponsesMessage{ + { + Role: schemas.Ptr(schemas.ResponsesInputMessageRoleUser), + Content: &schemas.ResponsesMessageContent{ + ContentStr: schemas.Ptr("Reply with exactly: lifecycle-ok"), + }, + }, + }, + Params: &schemas.ResponsesParameters{ + Store: &store, + }, + } + + created, err := client.ResponsesRequest(bfCtx, createReq) + if err != nil { + t.Fatalf("create stored response: %v", err) + } + if created == nil || created.ID == nil || *created.ID == "" { + t.Fatalf("expected non-empty response id") + } + rid := *created.ID + t.Cleanup(func() { + _, _ = client.ResponsesDeleteRequest(bfCtx, &schemas.BifrostResponsesDeleteRequest{ + Provider: testConfig.Provider, + ResponseID: rid, + }) + }) + + retrieved, err := client.ResponsesRetrieveRequest(bfCtx, &schemas.BifrostResponsesRetrieveRequest{ + Provider: testConfig.Provider, + ResponseID: rid, + }) + if err != nil { + t.Fatalf("retrieve: %v", err) + } + if retrieved == nil || retrieved.ID == nil || *retrieved.ID != rid { + t.Fatalf("retrieve id mismatch: got %#v want id %s", retrieved, rid) + } + + items, err := client.ResponsesInputItemsRequest(bfCtx, &schemas.BifrostResponsesInputItemsRequest{ + Provider: testConfig.Provider, + ResponseID: rid, + Limit: schemas.Ptr(20), + }) + if err != nil { + t.Fatalf("input_items: %v", err) + } + if items == nil || items.Object == "" { + t.Fatalf("expected input_items list payload") + } + + deleted, err := client.ResponsesDeleteRequest(bfCtx, &schemas.BifrostResponsesDeleteRequest{ + Provider: testConfig.Provider, + ResponseID: rid, + }) + if err != nil { + t.Fatalf("delete: %v", err) + } + if deleted == nil || !deleted.Deleted { + t.Fatalf("expected deleted response with deleted=true, got %#v", deleted) + } +} diff --git a/core/internal/llmtests/tests.go b/core/internal/llmtests/tests.go index 661280f0a2e..129bfded0b4 100644 --- a/core/internal/llmtests/tests.go +++ b/core/internal/llmtests/tests.go @@ -95,6 +95,7 @@ func RunAllComprehensiveTests(t *testing.T, client *bifrost.Bifrost, ctx context RunFileUnsupportedTest, RunFileAndBatchIntegrationTest, RunCountTokenTest, + RunResponsesLifecycleTest, RunChatAudioTest, RunChatAudioStreamTest, RunStructuredOutputChatTest, @@ -218,6 +219,7 @@ func printTestSummary(t *testing.T, testConfig ComprehensiveTestConfig) { {"FileUnsupported", !testConfig.Scenarios.FileUpload && !testConfig.Scenarios.FileList && !testConfig.Scenarios.FileRetrieve && !testConfig.Scenarios.FileDelete && !testConfig.Scenarios.FileContent}, {"FileAndBatchIntegration", testConfig.Scenarios.FileBatchInput}, {"CountTokens", testConfig.Scenarios.CountTokens}, + {"ResponsesLifecycle", testConfig.Scenarios.ResponsesLifecycle && testConfig.Provider == schemas.OpenAI}, {"ChatAudio", testConfig.Scenarios.ChatAudio && testConfig.ChatAudioModel != ""}, {"ChatAudioStream", testConfig.Scenarios.ChatAudio && testConfig.ChatAudioModel != ""}, {"StructuredOutputChat", testConfig.Scenarios.StructuredOutputs}, diff --git a/core/internal/llmtests/validation_presets.go b/core/internal/llmtests/validation_presets.go index 052cfc2c06a..0eeda8330aa 100644 --- a/core/internal/llmtests/validation_presets.go +++ b/core/internal/llmtests/validation_presets.go @@ -482,6 +482,10 @@ func ModifyExpectationsForProvider(expectations ResponseExpectations, provider s expectations.ShouldHaveUsageStats = true expectations.ShouldHaveLatency = true + case schemas.DeepSeek: + expectations.ShouldHaveUsageStats = true + expectations.ShouldHaveLatency = true + case schemas.OpenRouter: // OpenRouter proxies to multiple providers; returns OpenAI-compatible fields expectations.ShouldHaveUsageStats = true diff --git a/core/internal/mcptests/connect_ping_listtools_test.go b/core/internal/mcptests/connect_ping_listtools_test.go index 7058a40b301..f42f900c178 100644 --- a/core/internal/mcptests/connect_ping_listtools_test.go +++ b/core/internal/mcptests/connect_ping_listtools_test.go @@ -363,30 +363,40 @@ func TestListToolsHook_PreHookShortCircuitWithSyntheticTools(t *testing.T) { assert.False(t, hasEcho, "real server tools should not appear when PreHook short-circuited") } -func TestListToolsHook_PreHookShortCircuitError_LeavesEmptyToolMap(t *testing.T) { +// TestListToolsHook_InitialListToolsFailure_FailsConnect is the regression test for +// issue #4314. A transport that connects+initializes but whose initial list_tools call +// fails (here forced via a plugin error short-circuit) must NOT be registered as +// Connected with an empty ToolMap. Previously the connect path swallowed the failure +// and left a sticky "connected but serves zero tools" client that /api/mcp/clients +// reported as healthy while tools/list returned nothing. The fix treats it as a +// connection failure so the standard Disconnected + health-monitor reconnect path +// retries a full connect+list. +func TestListToolsHook_InitialListToolsFailure_FailsConnect(t *testing.T) { t.Parallel() plugin := NewTestListToolsPlugin() plugin.SetShortCircuitError(&schemas.BifrostError{ IsBifrostError: false, - Error: &schemas.ErrorField{Message: "list_tools blocked"}, + Error: &schemas.ErrorField{Message: "list_tools upstream timeout"}, }) manager, _ := setupBifrostWithPlugins(t, []schemas.MCPPlugin{plugin}) - // AddClient should still succeed — the connect path tolerates list_tools failure - // and falls back to empty tools (matching pre-plugin behavior). - require.NoError(t, manager.AddClient(context.Background(), inProcessClientConfig("list_err", buildInProcessServer(t)))) - clients := manager.GetClients() - var target *schemas.MCPClientState - for i := range clients { - if clients[i].Name == "list_err" { - target = &clients[i] - break - } + err := manager.AddClient(context.Background(), inProcessClientConfig("list_err", buildInProcessServer(t))) + require.Error(t, err, "AddClient must fail when initial list_tools errors") + assert.Contains(t, err.Error(), "list_tools upstream timeout") + + // AddClient cleans up the failed entry, so the client must not linger as a + // Connected-but-empty entry that would lie to /api/mcp/clients. + clientNames := make([]string, 0, len(manager.GetClients())) + for _, c := range manager.GetClients() { + clientNames = append(clientNames, c.Name) } - require.NotNil(t, target) - assert.Empty(t, target.ToolMap, "list_tools error short-circuit should result in empty ToolMap") + assert.NotContains(t, clientNames, "list_err", "failed list_tools client must be removed") + + // tools/list (served via GetToolPerClient) must not surface this client's tools. + served := manager.GetToolPerClient(context.Background()) + assert.Empty(t, served["list_err"], "no tools should be served for a client that failed list_tools") } func TestListToolsHook_FiresOnConnectAndAgain(t *testing.T) { diff --git a/core/internal/mcptests/per_server_timeout_test.go b/core/internal/mcptests/per_server_timeout_test.go new file mode 100644 index 00000000000..ab5fd4479e5 --- /dev/null +++ b/core/internal/mcptests/per_server_timeout_test.go @@ -0,0 +1,173 @@ +package mcptests + +import ( + "context" + "encoding/json" + "testing" + "time" + + mcpgo "github.com/mark3labs/mcp-go/mcp" + "github.com/mark3labs/mcp-go/server" + "github.com/maximhq/bifrost/core/schemas" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// buildPerServerDelayServer creates an InProcess MCP server with a delay tool +// that respects context cancellation so per-server timeouts are observable. +func buildPerServerDelayServer(t *testing.T) *server.MCPServer { + t.Helper() + s := server.NewMCPServer("delay-server", "1.0.0", server.WithToolCapabilities(true)) + delayTool := mcpgo.NewTool("delay", + mcpgo.WithDescription("Sleeps for the given number of seconds, respects context cancellation"), + mcpgo.WithNumber("seconds", mcpgo.Required(), mcpgo.Description("seconds to sleep")), + ) + s.AddTool(delayTool, func(ctx context.Context, req mcpgo.CallToolRequest) (*mcpgo.CallToolResult, error) { + seconds, _ := req.GetArguments()["seconds"].(float64) + timer := time.NewTimer(time.Duration(seconds * float64(time.Second))) + defer timer.Stop() + select { + case <-timer.C: + return mcpgo.NewToolResultText("ok"), nil + case <-ctx.Done(): + return mcpgo.NewToolResultError("timed out"), nil + } + }) + return s +} + +// makeDelayToolCall builds a ChatAssistantMessageToolCall for the delay tool on the named client. +func makeDelayToolCall(clientName string, seconds float64) schemas.ChatAssistantMessageToolCall { + args, _ := json.Marshal(map[string]interface{}{"seconds": seconds}) + toolName := clientName + "-delay" + return schemas.ChatAssistantMessageToolCall{ + ID: schemas.Ptr("call-delay"), + Type: schemas.Ptr("function"), + Function: schemas.ChatAssistantMessageToolCallFunction{ + Name: &toolName, + Arguments: string(args), + }, + } +} + +// TestPerServerTimeout_OverridesGlobal verifies that a per-server ToolExecutionTimeout +// fires before the global timeout and before the tool naturally finishes. +// Setup: per-server = 1s, tool delay = 3s, context = 10s. +// Expected: call returns in < 2.5s (per-server timeout fires at ~1s). +func TestPerServerTimeout_OverridesGlobal(t *testing.T) { + t.Parallel() + + clientName := "tsoverride" + cfg := &schemas.MCPClientConfig{ + ID: clientName + "-id", + Name: clientName, + ConnectionType: schemas.MCPConnectionTypeInProcess, + InProcessServer: buildPerServerDelayServer(t), + ToolsToExecute: []string{"*"}, + ToolExecutionTimeout: 1 * time.Second, + } + + manager := setupMCPManager(t) + require.NoError(t, manager.AddClient(context.Background(), cfg)) + + bf := setupBifrost(t) + bf.SetMCPManager(manager) + + ctx, cancel := createTestContextWithTimeout(10 * time.Second) + defer cancel() + + start := time.Now() + toolCall := makeDelayToolCall(clientName, 3.0) + result, bifrostErr := bf.ExecuteChatMCPTool(ctx, &toolCall) + elapsed := time.Since(start) + + assert.Less(t, elapsed, 2500*time.Millisecond, + "per-server timeout (1s) should fire before the 3s tool delay; got %v", elapsed) + // Timeout surfaces either as a BifrostError or as an error message in the result content. + if bifrostErr != nil && bifrostErr.Error != nil { + t.Logf("elapsed: %v, error: %s", elapsed, bifrostErr.Error.Message) + } else if result != nil { + t.Logf("elapsed: %v, timeout returned in result", elapsed) + } else { + t.Errorf("expected either a bifrost error or a result indicating timeout, got both nil") + } +} + +// TestPerServerTimeout_AllowsLongerThanGlobal verifies that when the per-server +// timeout is longer than the tool's execution time the tool completes successfully. +// Setup: per-server = 3s, tool delay = 2s, context = 10s. +// Expected: call succeeds with no error. +func TestPerServerTimeout_AllowsLongerThanGlobal(t *testing.T) { + t.Parallel() + + clientName := "tsallowslong" + cfg := &schemas.MCPClientConfig{ + ID: clientName + "-id", + Name: clientName, + ConnectionType: schemas.MCPConnectionTypeInProcess, + InProcessServer: buildPerServerDelayServer(t), + ToolsToExecute: []string{"*"}, + ToolExecutionTimeout: 3 * time.Second, + } + + manager := setupMCPManager(t) + require.NoError(t, manager.AddClient(context.Background(), cfg)) + + bf := setupBifrost(t) + bf.SetMCPManager(manager) + + ctx, cancel := createTestContextWithTimeout(10 * time.Second) + defer cancel() + + toolCall := makeDelayToolCall(clientName, 2.0) // 2s delay < 3s timeout → must succeed + result, bifrostErr := bf.ExecuteChatMCPTool(ctx, &toolCall) + + require.Nil(t, bifrostErr, "tool should succeed when delay < per-server timeout") + assert.NotNil(t, result, "should have a result") +} + +// TestPerServerTimeout_FallsBackToGlobal verifies that a client with +// ToolExecutionTimeout = 0 falls back to the global timeout / context deadline. +// A short context (500 ms) is used to make the timeout observable without +// depending on the manager's exact configured global value. +// Setup: per-server = 0 (use global), tool delay = 5s, context = 500ms. +// Expected: call returns well under 2s (context fires). +func TestPerServerTimeout_FallsBackToGlobal(t *testing.T) { + t.Parallel() + + clientName := "tsfallback" + cfg := &schemas.MCPClientConfig{ + ID: clientName + "-id", + Name: clientName, + ConnectionType: schemas.MCPConnectionTypeInProcess, + InProcessServer: buildPerServerDelayServer(t), + ToolsToExecute: []string{"*"}, + ToolExecutionTimeout: 0, // 0 = use global + } + + manager := setupMCPManager(t) + require.NoError(t, manager.AddClient(context.Background(), cfg)) + + bf := setupBifrost(t) + bf.SetMCPManager(manager) + + // Short context deadline — demonstrates that global/context applies when per-server = 0. + ctx, cancel := createTestContextWithTimeout(500 * time.Millisecond) + defer cancel() + + start := time.Now() + toolCall := makeDelayToolCall(clientName, 5.0) // 5s delay + result, bifrostErr := bf.ExecuteChatMCPTool(ctx, &toolCall) + elapsed := time.Since(start) + + assert.Less(t, elapsed, 2*time.Second, + "context deadline should cancel the tool (per-server=0 falls back to global); got %v", elapsed) + // Cancellation surfaces either as a BifrostError or as an error message in the result content. + if bifrostErr != nil && bifrostErr.Error != nil { + t.Logf("elapsed: %v, error: %s", elapsed, bifrostErr.Error.Message) + } else if result != nil { + t.Logf("elapsed: %v, cancellation returned in result", elapsed) + } else { + t.Errorf("expected either a bifrost error or a result indicating cancellation, got both nil") + } +} diff --git a/core/mcp/clientmanager.go b/core/mcp/clientmanager.go index 6eff9dc42ae..16109c13639 100644 --- a/core/mcp/clientmanager.go +++ b/core/mcp/clientmanager.go @@ -982,6 +982,7 @@ func (m *MCPManager) UpdateClient(id string, updatedConfig *schemas.MCPClientCon AllowedExtraHeaders: slices.Clone(updatedConfig.AllowedExtraHeaders), IsPingAvailable: updatedConfig.IsPingAvailable, ToolSyncInterval: updatedConfig.ToolSyncInterval, + ToolExecutionTimeout: updatedConfig.ToolExecutionTimeout, AllowOnAllVirtualKeys: updatedConfig.AllowOnAllVirtualKeys, Disabled: updatedConfig.Disabled, TLSConfig: updatedConfig.TLSConfig, @@ -1463,6 +1464,7 @@ func (m *MCPManager) connectToMCPClient(requestCtx context.Context, config *sche tools := make(map[string]schemas.ChatTool) toolNameMapping := make(map[string]string) + var listToolsErr error if externalClient == nil { // Plugin short-circuited the connect with a success response; no live transport // to query. Register the client as "connected" with an empty tool set — this is @@ -1481,8 +1483,8 @@ func (m *MCPManager) connectToMCPClient(requestCtx context.Context, config *sche defer toolRetrievalCancel() t, mapping, err := m.runListToolsWithHooks(toolRetrievalCtx, externalClient, config.Name) if err != nil { + listToolsErr = err m.logger.Warn("%s Failed to retrieve tools from %s: %v", MCPLogPrefix, config.Name, err) - // Continue with connection even if tool retrieval fails } else { tools = t toolNameMapping = mapping @@ -1490,6 +1492,24 @@ func (m *MCPManager) connectToMCPClient(requestCtx context.Context, config *sche m.logger.Debug("%s [%s] Retrieved %d tools", MCPLogPrefix, config.Name, len(tools)) } + // A live transport that cannot enumerate its tools is not healthy. Marking it + // Connected with an empty ToolMap makes /api/mcp/clients disagree with tools/list + // and is sticky — ping-based health keeps succeeding (transport is alive), so a + // reconnect that would re-run discovery never fires. Treat a failed initial + // list_tools as a connection failure: tear down the transport and return an error + // so the standard Disconnected + health-monitor reconnect path retries a full + // connect+list. A server that legitimately exposes zero tools returns success with + // an empty list (listToolsErr == nil) and is still marked Connected. + if listToolsErr != nil { + if (config.ConnectionType == schemas.MCPConnectionTypeSSE || config.ConnectionType == schemas.MCPConnectionTypeSTDIO) && cancel != nil { + cancel() + } + if closeErr := externalClient.Close(); closeErr != nil { + m.logger.Warn("%s Failed to close external client after tool retrieval failure: %v", MCPLogPrefix, closeErr) + } + return fmt.Errorf("failed to retrieve tools from MCP client %s: %w", config.Name, listToolsErr) + } + // Second lock: Update client with final connection details and tools m.mu.Lock() diff --git a/core/mcp/credstore/per_user_headers.go b/core/mcp/credstore/per_user_headers.go index 70f0dc7dca1..fa1b5fd56e9 100644 --- a/core/mcp/credstore/per_user_headers.go +++ b/core/mcp/credstore/per_user_headers.go @@ -88,6 +88,11 @@ func (r *perUserHeadersResolver) buildAuthRequiredError(ctx *schemas.BifrostCont // can't accidentally start flow rows with empty identity columns. return fmt.Errorf("per-user headers auth-required flow requires an identity") } + // No identity gate here (see per_user_oauth.go for the rationale). For + // user-mode flows the submission is verified at the cookie-bearing UI step + // (flowSubmit → canAccessUserFlow requires the dashboard user to match + // flow.UserID, and user-mode flows mint no shareable temp token); the flow + // row created here grants nothing on its own. initiation, err := r.provider.InitiateUserSubmissionFlow(ctx, mode, identity, config.ID, baseURL) if err != nil { return fmt.Errorf("failed to initiate per-user headers submission flow for %s: %w", config.Name, err) diff --git a/core/mcp/credstore/per_user_oauth.go b/core/mcp/credstore/per_user_oauth.go index 41ddb6bfd8c..44ba851ad50 100644 --- a/core/mcp/credstore/per_user_oauth.go +++ b/core/mcp/credstore/per_user_oauth.go @@ -57,6 +57,13 @@ func (r *perUserOAuthResolver) ConnectionHeaders(ctx *schemas.BifrostContext, co if redirectURI == "" { return nil, fmt.Errorf("per-user OAuth requires a redirect URI but none is available in context") } + // No identity gate here. A user-mode caller (e.g. an MCP client presenting + // only a Bearer JWT) carries no dashboard session at tool-call time. The + // flow row records the caller's identity (flow.UserID = bf_sub); the binding + // is verified at the cookie-bearing UI step (flowStart → canAccessUserFlow) + // before the upstream authorize URL — which carries the single-use state — is + // ever revealed, and the callback binds the token to the flow's recorded + // identity. So initiating the flow here grants nothing on its own. flowInitiation, sessionID, flowErr := r.provider.InitiateUserOAuthFlow(ctx, *config.OauthConfigID, config.ID, redirectURI, mode) if flowErr != nil { return nil, fmt.Errorf("failed to initiate per-user OAuth flow for %s: %w", config.Name, flowErr) diff --git a/core/mcp/toolmanager.go b/core/mcp/toolmanager.go index 733343c270a..c24cad6df8d 100644 --- a/core/mcp/toolmanager.go +++ b/core/mcp/toolmanager.go @@ -683,8 +683,12 @@ func (m *ToolsManager) executeToolInternal( sanitizedToolName := stripClientPrefix(toolName, executionConfig.Name) originalMCPToolName := getOriginalToolName(sanitizedToolName, toolNameMapping) - // Create timeout context for tool execution + // Create timeout context for tool execution. + // Per-server timeout (executionConfig.ToolExecutionTimeout) takes precedence over the global. toolExecutionTimeout := m.toolExecutionTimeout.Load().(time.Duration) + if executionConfig != nil && executionConfig.ToolExecutionTimeout > 0 { + toolExecutionTimeout = executionConfig.ToolExecutionTimeout + } toolCtx, cancel := context.WithTimeout(ctx, toolExecutionTimeout) defer cancel() diff --git a/core/network/dialaddrhost_test.go b/core/network/dialaddrhost_test.go new file mode 100644 index 00000000000..8af2087efab --- /dev/null +++ b/core/network/dialaddrhost_test.go @@ -0,0 +1,32 @@ +package network + +import "testing" + +func TestDialAddrHost(t *testing.T) { + tests := []struct { + addr string + want string + }{ + {"example.com:443", "example.com"}, + {"127.0.0.1:8080", "127.0.0.1"}, + {"[::1]:8080", "::1"}, + {"[2001:db8::1]:443", "2001:db8::1"}, + {"example.com", "example.com"}, // no port + {"[::1]", "::1"}, // bracketed, no port + } + for _, tt := range tests { + if got := dialAddrHost(tt.addr); got != tt.want { + t.Errorf("dialAddrHost(%q) = %q, want %q", tt.addr, got, tt.want) + } + } +} + +func TestNoProxyBypassIPv6(t *testing.T) { + // An IPv6 literal listed in no_proxy must match after host extraction + if !shouldBypassProxy(dialAddrHost("[::1]:8080"), "::1") { + t.Error("[::1]:8080 should bypass proxy when no_proxy contains ::1") + } + if shouldBypassProxy(dialAddrHost("[2001:db8::1]:443"), "::1") { + t.Error("non-listed IPv6 target must not bypass proxy") + } +} diff --git a/core/network/http.go b/core/network/http.go index 89d7aaf83f2..de4ba729cf5 100644 --- a/core/network/http.go +++ b/core/network/http.go @@ -322,11 +322,7 @@ func (f *HTTPClientFactory) configureFasthttpProxy(client *fasthttp.Client) { if dialFunc != nil { client.Dial = func(addr string) (net.Conn, error) { if proxyCfg.NoProxy != "" { - host := strings.Split(addr, ":")[0] - if host == "" { - host = addr - } - if shouldBypassProxy(host, proxyCfg.NoProxy) { + if shouldBypassProxy(dialAddrHost(addr), proxyCfg.NoProxy) { return net.Dial("tcp", addr) } } @@ -335,6 +331,17 @@ func (f *HTTPClientFactory) configureFasthttpProxy(client *fasthttp.Client) { } } +// dialAddrHost extracts the host from a dial target for no_proxy matching. +// SplitHostPort unwraps IPv6 brackets ("[::1]:8080" -> "::1"); naive splitting +// on ":" would mangle IPv6 literals. +func dialAddrHost(addr string) string { + host, _, err := net.SplitHostPort(addr) + if err != nil || host == "" { + host = strings.Trim(addr, "[]") + } + return host +} + // createHTTPClient creates a new standard net/http client with appropriate proxy settings func (f *HTTPClientFactory) createHTTPClient(purpose ClientPurpose) *http.Client { transport := &http.Transport{ diff --git a/core/providers/anthropic/advisor_test.go b/core/providers/anthropic/advisor_test.go index 73f25c8deac..a2f9723a544 100644 --- a/core/providers/anthropic/advisor_test.go +++ b/core/providers/anthropic/advisor_test.go @@ -404,8 +404,8 @@ func TestResponsesStream_MessageStart_NoDuplicateOnPassthrough(t *testing.T) { t.Fatalf("unmarshal message_start: %v", err) } - state := acquireAnthropicResponsesStreamState() - defer releaseAnthropicResponsesStreamState(state) + state := AcquireAnthropicResponsesStreamState() + defer ReleaseAnthropicResponsesStreamState(state) responses, bErr, _ := chunk.ToBifrostResponsesStream(ctx, 0, state) if bErr != nil { diff --git a/core/providers/anthropic/anthropic.go b/core/providers/anthropic/anthropic.go index 76e51365fcc..a5e44a272fc 100644 --- a/core/providers/anthropic/anthropic.go +++ b/core/providers/anthropic/anthropic.go @@ -9,6 +9,7 @@ import ( "io" "mime/multipart" "net/http" + "net/textproto" "net/url" "strings" "sync" @@ -171,13 +172,41 @@ func extractAnthropicResponsesUsageFromPrefetch(data []byte) *schemas.ResponsesR } } -// completeRequest sends a request to Anthropic's API and handles the response. -// It constructs the API URL, sets up authentication, and processes the response. -// Returns the response body or an error if the request fails. -// When large response streaming is activated (BifrostContextKeyLargeResponseMode set in ctx), -// returns (nil, latency, nil) — callers must check the context flag. -func (provider *AnthropicProvider) completeRequest(ctx *schemas.BifrostContext, jsonData []byte, url string, key string, requestType schemas.RequestType) ([]byte, time.Duration, map[string]string, *schemas.BifrostError) { - // Create the request with the JSON body +// anthropicRequestHeaders builds the auth/version headers for an Anthropic request: the API +// version plus x-api-key when a key is present and Claude Code max-mode is off. Shared by the +// provider's unary (completeRequest / HandleAnthropic*Request) and streaming paths. +func (provider *AnthropicProvider) anthropicRequestHeaders(ctx *schemas.BifrostContext, key schemas.Key) map[string]string { + headers := map[string]string{ + "anthropic-version": provider.apiVersion, + } + if key.Value.GetValue() != "" && !IsClaudeCodeMaxMode(ctx) { + headers["x-api-key"] = key.Value.GetValue() + } + return headers +} + +// completeRequest sends a non-streaming Anthropic Messages API request over fasthttp and +// returns the buffered response body, latency, and provider response headers. It is the single +// unary request core for the package: the exported HandleAnthropic*Request handlers call it +// (then parse), and the provider's TextCompletion / CountTokens paths call it directly. It +// mirrors the request shaping of HandleAnthropicChatCompletionStreaming (header application, +// beta filtering, large-payload request body, large-response detection) but reads a full +// response instead of a stream. Auth and version headers are supplied by the caller via the +// headers map. On a large response it returns a nil body and signals via the +// BifrostContextKeyLargeResponseMode context value (set by FinalizeResponseWithLargeDetection). +func completeRequest( + ctx *schemas.BifrostContext, + client *fasthttp.Client, + url string, + jsonBody []byte, + headers map[string]string, + extraHeaders map[string]string, + betaHeaderOverrides map[string]bool, + providerName schemas.ModelProvider, + requestType schemas.RequestType, + signer providerUtils.BodySigner, + logger schemas.Logger, +) ([]byte, time.Duration, map[string]string, *schemas.BifrostError) { req := fasthttp.AcquireRequest() resp := fasthttp.AcquireResponse() defer fasthttp.ReleaseRequest(req) @@ -188,35 +217,49 @@ func (provider *AnthropicProvider) completeRequest(ctx *schemas.BifrostContext, } }() - // Set any extra headers from network config - providerUtils.SetExtraHeaders(ctx, req, provider.networkConfig.ExtraHeaders, nil) req.SetRequestURI(url) req.Header.SetMethod(http.MethodPost) - req.Header.SetContentType("application/json") - // Can be empty in case of passthrough or keyless custom provider - // Here we can avoid this - in case of passthrough completely - if key != "" && !IsClaudeCodeMaxMode(ctx) { - req.Header.Set("x-api-key", key) - } - req.Header.Set("anthropic-version", provider.apiVersion) + // Set network-config extra headers, excluding anthropic-beta (set explicitly below). + providerUtils.SetExtraHeaders(ctx, req, extraHeaders, []string{AnthropicBetaHeader}) - if betaHeaders := FilterBetaHeadersForProvider(MergeBetaHeaders(ctx, provider.networkConfig.ExtraHeaders), schemas.Anthropic, provider.networkConfig.BetaHeaderOverrides); len(betaHeaders) > 0 { + // Force JSON content type after extra headers so network config can't override it + // (matches the original per-provider completeRequest ordering). + req.Header.SetContentType("application/json") + + if betaHeaders := FilterBetaHeadersForProvider(MergeBetaHeaders(ctx, extraHeaders), providerName, betaHeaderOverrides); len(betaHeaders) > 0 { req.Header.Set(AnthropicBetaHeader, strings.Join(betaHeaders, ",")) } else { req.Header.Del(AnthropicBetaHeader) } - usedLargePayloadBody := setAnthropicRequestBody(ctx, req, jsonData) + // Apply caller-supplied auth/version headers last so they win over network-config headers. + for key, value := range headers { + req.Header.Set(key, value) + } + + usedLargePayloadBody := setAnthropicRequestBody(ctx, req, jsonBody) - requestClient := provider.client + // Sign the exact body bytes set on the request, when a signer is supplied (e.g. AWS SigV4 + // for Bedrock Mantle). Done after the body is set so the signature covers what is actually sent. + if signer != nil { + sigHeaders, bErr := signer(jsonBody) + if bErr != nil { + return nil, 0, nil, bErr + } + for k, v := range sigHeaders { + req.Header.Set(k, v) + } + } + + requestClient := client responseThreshold, _ := ctx.Value(schemas.BifrostContextKeyLargeResponseThreshold).(int64) isCountTokens := requestType == schemas.CountTokensRequest - // CountTokens responses are always tiny — skip streaming client so the response - // is buffered normally (same approach as OpenAI and Gemini count_tokens handlers). + // Count-tokens responses are always tiny — skip the large-response streaming client so the + // response is buffered normally. if responseThreshold > 0 && !isCountTokens { resp.StreamBody = true - requestClient = providerUtils.BuildLargeResponseClient(provider.client, responseThreshold) + requestClient = providerUtils.BuildLargeResponseClient(client, responseThreshold) } // Send the request @@ -235,11 +278,11 @@ func (provider *AnthropicProvider) completeRequest(ctx *schemas.BifrostContext, // Handle error response — materialize stream body for error parsing if resp.StatusCode() != fasthttp.StatusOK { providerUtils.MaterializeStreamErrorBody(ctx, resp) - provider.logger.Debug("error from %s provider: %s", provider.GetProviderKey(), string(resp.Body())) - return nil, latency, providerResponseHeaders, parseAnthropicError(resp) + logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) + return nil, latency, providerResponseHeaders, providerUtils.SetErrorLatency(parseAnthropicError(resp), latency) } - // CountTokens uses buffered response (streaming skipped above) — decode directly. + // Count-tokens uses a buffered response (large-response streaming skipped above). if isCountTokens { body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { @@ -249,7 +292,7 @@ func (provider *AnthropicProvider) completeRequest(ctx *schemas.BifrostContext, } // Delegate large response detection + normal buffered path to shared utility - body, isLarge, respErr := providerUtils.FinalizeResponseWithLargeDetection(ctx, resp, provider.logger) + body, isLarge, respErr := providerUtils.FinalizeResponseWithLargeDetection(ctx, resp, logger) if respErr != nil { return nil, latency, providerResponseHeaders, respErr } @@ -293,7 +336,7 @@ func (provider *AnthropicProvider) listModelsByKey(ctx *schemas.BifrostContext, // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, parseAnthropicError(resp) + return nil, providerUtils.SetErrorLatency(parseAnthropicError(resp), latency) } // Parse Anthropic's response @@ -361,12 +404,12 @@ func (provider *AnthropicProvider) TextCompletion(ctx *schemas.BifrostContext, k } // Use struct directly for JSON marshaling (no beta headers for text completion) - responseBody, latency, providerResponseHeaders, err := provider.completeRequest(ctx, jsonData, provider.buildRequestURL(ctx, "/v1/complete", schemas.TextCompletionRequest), key.Value.GetValue(), schemas.TextCompletionRequest) + responseBody, latency, providerResponseHeaders, err := completeRequest(ctx, provider.client, provider.buildRequestURL(ctx, "/v1/complete", schemas.TextCompletionRequest), jsonData, provider.anthropicRequestHeaders(ctx, key), provider.networkConfig.ExtraHeaders, provider.networkConfig.BetaHeaderOverrides, provider.GetProviderKey(), schemas.TextCompletionRequest, nil, provider.logger) if providerResponseHeaders != nil { ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Large response mode: return lightweight response with metadata only @@ -386,7 +429,7 @@ func (provider *AnthropicProvider) TextCompletion(ctx *schemas.BifrostContext, k rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, response, jsonData, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse := response.ToBifrostTextCompletionResponse() @@ -422,54 +465,58 @@ func (provider *AnthropicProvider) ChatCompletion(ctx *schemas.BifrostContext, k if err := providerUtils.CheckOperationAllowed(schemas.Anthropic, provider.customProviderConfig, schemas.ChatCompletionRequest); err != nil { return nil, err } - // Convert to Anthropic format and get required beta headers - jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody( + return HandleAnthropicChatCompletionRequest( ctx, + provider.client, + provider.buildRequestURL(ctx, "/v1/messages", schemas.ChatCompletionRequest), request, - func() (providerUtils.RequestBodyWithExtraParams, error) { - anthropicReq, convErr := ToAnthropicChatRequest(ctx, request) - if convErr != nil { - return nil, convErr - } - AddMissingBetaHeadersToContext(ctx, anthropicReq, schemas.Anthropic) - return anthropicReq, nil - }) + AnthropicRequestBuildConfig{ + Provider: schemas.Anthropic, + IsStreaming: false, + BetaHeaderOverrides: provider.networkConfig.BetaHeaderOverrides, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }, + provider.anthropicRequestHeaders(ctx, key), + provider.networkConfig.ExtraHeaders, + nil, + provider.logger, + ) +} + +// HandleAnthropicChatCompletionRequest builds the Anthropic Messages chat request body from +// config, performs a non-streaming request, and parses the native Anthropic response into a +// BifrostChatResponse. It is the unary counterpart to HandleAnthropicChatCompletionStreaming, +// shared by the Anthropic, Azure, Vertex, and Bedrock providers. Callers supply the request, +// the per-provider build config (Provider, Model, ShouldSendBackRaw*, BetaHeaderOverrides), +// the request URL, and the auth/version headers (x-api-key, Bearer, SigV4, anthropic-version) +// via the headers map; beta headers are filtered for config.Provider. +func HandleAnthropicChatCompletionRequest( + ctx *schemas.BifrostContext, + client *fasthttp.Client, + url string, + request *schemas.BifrostChatRequest, + config AnthropicRequestBuildConfig, + headers map[string]string, + extraHeaders map[string]string, + signer providerUtils.BodySigner, + logger schemas.Logger, +) (*schemas.BifrostChatResponse, *schemas.BifrostError) { + jsonBody, bifrostErr := BuildAnthropicChatRequestBody(ctx, request, config) if bifrostErr != nil { return nil, bifrostErr } - // On the raw-body passthrough path, the typed-struct StripUnsupportedAnthropicFields - // was not invoked. Apply the JSON-level sanitizer for behavioural parity so - // unsupported request-level and tool-level fields don't leak to providers that - // would reject them. - if useRawBody, ok := ctx.Value(schemas.BifrostContextKeyUseRawRequestBody).(bool); ok && useRawBody { - // Feature gating keyed to schemas.Anthropic (not provider.GetProviderKey()) - // so custom Anthropic aliases get the same feature lookup as the typed - // path above (line 445), keeping raw and typed behavior in lockstep. - sanitized, rawErr := StripUnsupportedFieldsFromRawBody(jsonData, schemas.Anthropic, schemas.ResolveCanonicalModel(ctx, request.Model)) - if rawErr != nil { - return nil, providerUtils.NewBifrostOperationError(schemas.ErrProviderRequestMarshal, rawErr) - } - jsonData = sanitized - // Auto-inject matching anthropic-beta headers for fields the sanitizer - // preserved. Probe-unmarshal reuses the typed path's header walker so - // the two paths stay in lockstep. - var probe AnthropicMessageRequest - if err := schemas.Unmarshal(jsonData, &probe); err == nil { - AddMissingBetaHeadersToContext(ctx, &probe, schemas.Anthropic) - } - } - // Use struct directly for JSON marshaling - responseBody, latency, providerResponseHeaders, err := provider.completeRequest(ctx, jsonData, provider.buildRequestURL(ctx, "/v1/messages", schemas.ChatCompletionRequest), key.Value.GetValue(), schemas.ChatCompletionRequest) + responseBody, latency, providerResponseHeaders, bifrostErr := completeRequest(ctx, client, url, jsonBody, headers, extraHeaders, config.BetaHeaderOverrides, config.Provider, schemas.ChatCompletionRequest, signer, logger) if providerResponseHeaders != nil { ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } - if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + if bifrostErr != nil { + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, config.ShouldSendBackRawRequest, config.ShouldSendBackRawResponse, latency) } - // Large response mode: return lightweight response with metadata only + // Large response mode: return lightweight response with metadata only. if isLargeResp, _ := ctx.Value(schemas.BifrostContextKeyLargeResponseMode).(bool); isLargeResp { return &schemas.BifrostChatResponse{ Model: request.Model, @@ -484,9 +531,9 @@ func (provider *AnthropicProvider) ChatCompletion(ctx *schemas.BifrostContext, k response := AcquireAnthropicMessageResponse() defer ReleaseAnthropicMessageResponse(response) - rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, response, jsonData, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) + rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, response, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, config.ShouldSendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, config.ShouldSendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, config.ShouldSendBackRawRequest, config.ShouldSendBackRawResponse, latency) } // Create final response bifrostResponse := response.ToBifrostChatResponse(ctx) @@ -494,17 +541,14 @@ func (provider *AnthropicProvider) ChatCompletion(ctx *schemas.BifrostContext, k // Set ExtraFields bifrostResponse.ExtraFields.Latency = latency.Milliseconds() bifrostResponse.ExtraFields.ProviderResponseHeaders = providerResponseHeaders - // Set raw request if enabled - if providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest) { + if providerUtils.ShouldSendBackRawRequest(ctx, config.ShouldSendBackRawRequest) { bifrostResponse.ExtraFields.RawRequest = rawRequest } - // Set raw response if enabled - if providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) { + if providerUtils.ShouldSendBackRawResponse(ctx, config.ShouldSendBackRawResponse) { bifrostResponse.ExtraFields.RawResponse = rawResponse } - return bifrostResponse, nil } @@ -516,42 +560,16 @@ func (provider *AnthropicProvider) ChatCompletionStream(ctx *schemas.BifrostCont return nil, err } - // Convert to Anthropic format and get required beta headers - jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody( - ctx, - request, - func() (providerUtils.RequestBodyWithExtraParams, error) { - anthropicReq, convErr := ToAnthropicChatRequest(ctx, request) - if convErr != nil { - return nil, convErr - } - anthropicReq.Stream = schemas.Ptr(true) - AddMissingBetaHeadersToContext(ctx, anthropicReq, schemas.Anthropic) - return anthropicReq, nil - }) + jsonData, bifrostErr := BuildAnthropicChatRequestBody(ctx, request, AnthropicRequestBuildConfig{ + Provider: schemas.Anthropic, + IsStreaming: true, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }) if bifrostErr != nil { return nil, bifrostErr } - // On the raw-body passthrough path, the typed-struct StripUnsupportedAnthropicFields - // was not invoked. Apply the JSON-level sanitizer for behavioural parity. - if useRawBody, ok := ctx.Value(schemas.BifrostContextKeyUseRawRequestBody).(bool); ok && useRawBody { - // Feature gating keyed to schemas.Anthropic (not provider.GetProviderKey()) - // to keep raw and typed paths in lockstep on custom aliases — mirrors - // the typed path's hardcoded schemas.Anthropic at line 548. - sanitized, rawErr := StripUnsupportedFieldsFromRawBody(jsonData, schemas.Anthropic, schemas.ResolveCanonicalModel(ctx, request.Model)) - if rawErr != nil { - return nil, providerUtils.NewBifrostOperationError(schemas.ErrProviderRequestMarshal, rawErr) - } - jsonData = sanitized - // Auto-inject matching anthropic-beta headers for fields the sanitizer - // preserved. Probe-unmarshal reuses the typed path's header walker. - var probe AnthropicMessageRequest - if err := schemas.Unmarshal(jsonData, &probe); err == nil { - AddMissingBetaHeadersToContext(ctx, &probe, schemas.Anthropic) - } - } - // Prepare Anthropic headers headers := map[string]string{ "Content-Type": "application/json", @@ -579,6 +597,7 @@ func (provider *AnthropicProvider) ChatCompletionStream(ctx *schemas.BifrostCont provider.GetProviderKey(), postHookRunner, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -598,6 +617,81 @@ func normalizeCachedUsage(usage *schemas.BifrostLLMUsage) { usage.TotalTokens += cached } +func accumulateAnthropicResponsesUsage(usage *schemas.ResponsesResponseUsage, billedUsage *schemas.BifrostLLMUsage, usageToProcess *AnthropicUsage) { + if usage == nil || usageToProcess == nil { + return + } + if usageToProcess.InputTokens > usage.InputTokens { + usage.InputTokens = usageToProcess.InputTokens + if billedUsage != nil { + billedUsage.PromptTokens = usageToProcess.InputTokens + } + } + if usageToProcess.OutputTokens > usage.OutputTokens { + usage.OutputTokens = usageToProcess.OutputTokens + if billedUsage != nil { + billedUsage.CompletionTokens = usageToProcess.OutputTokens + } + } + calculatedTotal := usage.InputTokens + usage.OutputTokens + if calculatedTotal > usage.TotalTokens { + usage.TotalTokens = calculatedTotal + if billedUsage != nil { + billedUsage.TotalTokens = calculatedTotal + } + } + // Handle cached tokens if present + if usageToProcess.CacheReadInputTokens > 0 { + if usage.InputTokensDetails == nil { + usage.InputTokensDetails = &schemas.ResponsesResponseInputTokens{} + } + if billedUsage != nil && billedUsage.PromptTokensDetails == nil { + billedUsage.PromptTokensDetails = &schemas.ChatPromptTokensDetails{} + } + if usageToProcess.CacheReadInputTokens > usage.InputTokensDetails.CachedReadTokens { + usage.InputTokensDetails.CachedReadTokens = usageToProcess.CacheReadInputTokens + if billedUsage != nil { + billedUsage.PromptTokensDetails.CachedReadTokens = usageToProcess.CacheReadInputTokens + } + } + } + // Handle cached tokens if present + if usageToProcess.CacheCreationInputTokens > 0 { + if usage.InputTokensDetails == nil { + usage.InputTokensDetails = &schemas.ResponsesResponseInputTokens{} + } + if billedUsage != nil && billedUsage.PromptTokensDetails == nil { + billedUsage.PromptTokensDetails = &schemas.ChatPromptTokensDetails{} + } + if usageToProcess.CacheCreationInputTokens > usage.InputTokensDetails.CachedWriteTokens { + usage.InputTokensDetails.CachedWriteTokens = usageToProcess.CacheCreationInputTokens + if billedUsage != nil { + billedUsage.PromptTokensDetails.CachedWriteTokens = usageToProcess.CacheCreationInputTokens + } + } + if usageToProcess.CacheCreation.Ephemeral5mInputTokens > 0 || usageToProcess.CacheCreation.Ephemeral1hInputTokens > 0 { + if usage.InputTokensDetails.CachedWriteTokenDetails == nil { + usage.InputTokensDetails.CachedWriteTokenDetails = &schemas.ChatCachedWriteTokenDetails{} + } + if billedUsage != nil && billedUsage.PromptTokensDetails.CachedWriteTokenDetails == nil { + billedUsage.PromptTokensDetails.CachedWriteTokenDetails = &schemas.ChatCachedWriteTokenDetails{} + } + if usageToProcess.CacheCreation.Ephemeral5mInputTokens > usage.InputTokensDetails.CachedWriteTokenDetails.CachedWriteTokens5m { + usage.InputTokensDetails.CachedWriteTokenDetails.CachedWriteTokens5m = usageToProcess.CacheCreation.Ephemeral5mInputTokens + if billedUsage != nil { + billedUsage.PromptTokensDetails.CachedWriteTokenDetails.CachedWriteTokens5m = usageToProcess.CacheCreation.Ephemeral5mInputTokens + } + } + if usageToProcess.CacheCreation.Ephemeral1hInputTokens > usage.InputTokensDetails.CachedWriteTokenDetails.CachedWriteTokens1h { + usage.InputTokensDetails.CachedWriteTokenDetails.CachedWriteTokens1h = usageToProcess.CacheCreation.Ephemeral1hInputTokens + if billedUsage != nil { + billedUsage.PromptTokensDetails.CachedWriteTokenDetails.CachedWriteTokens1h = usageToProcess.CacheCreation.Ephemeral1hInputTokens + } + } + } + } +} + // HandleAnthropicChatCompletionStreaming handles streaming for Anthropic-compatible APIs. // This shared function reduces code duplication between providers that use the same SSE event format. func HandleAnthropicChatCompletionStreaming( @@ -614,6 +708,7 @@ func HandleAnthropicChatCompletionStreaming( providerName schemas.ModelProvider, postHookRunner schemas.PostHookRunner, postResponseConverter func(*schemas.BifrostChatResponse) *schemas.BifrostChatResponse, + signer providerUtils.BodySigner, logger schemas.Logger, postHookSpanFinalizer func(context.Context), ) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { @@ -641,6 +736,19 @@ func HandleAnthropicChatCompletionStreaming( usedLargePayloadBody := setAnthropicRequestBody(ctx, req, jsonBody) + // Sign the exact body bytes set on the request, when a signer is supplied (e.g. AWS SigV4 + // for Bedrock Mantle). Done after the body is set so the signature covers what is actually sent. + if signer != nil { + sigHeaders, bErr := signer(jsonBody) + if bErr != nil { + defer providerUtils.ReleaseStreamingResponse(ctx, resp) + return nil, bErr + } + for k, v := range sigHeaders { + req.Header.Set(k, v) + } + } + // Close the connection after this streaming response instead of returning it to // the keep-alive pool. fasthttp can otherwise reuse a streaming connection whose // reader is still in a torn state (body not fully drained, or the idle-timeout/ @@ -662,6 +770,7 @@ func HandleAnthropicChatCompletionStreaming( startTime := time.Now() // Make the request err := activeClient.Do(req, resp) + latency := time.Since(startTime) if usedLargePayloadBody { providerUtils.DrainLargePayloadRemainder(ctx) } @@ -675,16 +784,16 @@ func HandleAnthropicChatCompletionStreaming( Message: schemas.ErrRequestCancelled, Error: err, }, - }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Request failed before the first response byte (server closed an idle/pooled connection, // broken pipe, connection refused, DNS failure, etc.). Surface as a retriable upstream // connection error (502) so executeRequestWithRetries honors max_retries, matching the // non-streaming path - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Store provider response headers in context before status check so error responses also forward them @@ -693,7 +802,7 @@ func HandleAnthropicChatCompletionStreaming( // Check for HTTP errors if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) - return nil, providerUtils.EnrichError(ctx, parseAnthropicError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseAnthropicError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Large payload streaming passthrough — pipe raw upstream SSE to client @@ -725,7 +834,7 @@ func HandleAnthropicChatCompletionStreaming( fmt.Errorf("provider returned an empty response"), ) ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } @@ -1009,20 +1118,54 @@ func (provider *AnthropicProvider) Responses(ctx *schemas.BifrostContext, key sc if err := providerUtils.CheckOperationAllowed(schemas.Anthropic, provider.customProviderConfig, schemas.ResponsesRequest); err != nil { return nil, err } - jsonBody, err := getRequestBodyForResponses(ctx, request, false, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) - if err != nil { - return nil, err + return HandleAnthropicResponsesRequest( + ctx, + provider.client, + provider.buildRequestURL(ctx, "/v1/messages", schemas.ResponsesRequest), + request, + AnthropicRequestBuildConfig{ + Provider: schemas.Anthropic, + IsStreaming: false, + BetaHeaderOverrides: provider.networkConfig.BetaHeaderOverrides, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }, + provider.anthropicRequestHeaders(ctx, key), + provider.networkConfig.ExtraHeaders, + nil, + provider.logger, + ) +} + +// HandleAnthropicResponsesRequest is the Responses-API analogue of +// HandleAnthropicChatCompletionRequest: it builds the Anthropic Messages request body from +// config, performs a non-streaming request, and parses the native Anthropic response into a +// BifrostResponsesResponse. Shared by the Anthropic, Azure, Vertex, and Bedrock providers. +func HandleAnthropicResponsesRequest( + ctx *schemas.BifrostContext, + client *fasthttp.Client, + url string, + request *schemas.BifrostResponsesRequest, + config AnthropicRequestBuildConfig, + headers map[string]string, + extraHeaders map[string]string, + signer providerUtils.BodySigner, + logger schemas.Logger, +) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) { + jsonBody, bifrostErr := BuildAnthropicResponsesRequestBody(ctx, request, config) + if bifrostErr != nil { + return nil, bifrostErr } - responseBody, latency, providerResponseHeaders, err := provider.completeRequest(ctx, jsonBody, provider.buildRequestURL(ctx, "/v1/messages", schemas.ResponsesRequest), key.Value.GetValue(), schemas.ResponsesRequest) + responseBody, latency, providerResponseHeaders, bifrostErr := completeRequest(ctx, client, url, jsonBody, headers, extraHeaders, config.BetaHeaderOverrides, config.Provider, schemas.ResponsesRequest, signer, logger) if providerResponseHeaders != nil { ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } - if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + if bifrostErr != nil { + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, config.ShouldSendBackRawRequest, config.ShouldSendBackRawResponse, latency) } - // Large response mode: return lightweight response with usage from preview for plugin pipeline. + // Large response mode: return lightweight response with usage from the prefetch preview. if isLargeResp, _ := ctx.Value(schemas.BifrostContextKeyLargeResponseMode).(bool); isLargeResp { preview, _ := ctx.Value(schemas.BifrostContextKeyLargePayloadResponsePreview).(string) return &schemas.BifrostResponsesResponse{ @@ -1042,9 +1185,9 @@ func (provider *AnthropicProvider) Responses(ctx *schemas.BifrostContext, key sc response := AcquireAnthropicMessageResponse() defer ReleaseAnthropicMessageResponse(response) - rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, response, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) + rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, response, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, config.ShouldSendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, config.ShouldSendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, config.ShouldSendBackRawRequest, config.ShouldSendBackRawResponse, latency) } // Create final response @@ -1053,17 +1196,14 @@ func (provider *AnthropicProvider) Responses(ctx *schemas.BifrostContext, key sc // Set ExtraFields bifrostResponse.ExtraFields.Latency = latency.Milliseconds() bifrostResponse.ExtraFields.ProviderResponseHeaders = providerResponseHeaders - // Set raw request if enabled - if providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest) { + if providerUtils.ShouldSendBackRawRequest(ctx, config.ShouldSendBackRawRequest) { bifrostResponse.ExtraFields.RawRequest = rawRequest } - // Set raw response if enabled - if providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) { + if providerUtils.ShouldSendBackRawResponse(ctx, config.ShouldSendBackRawResponse) { bifrostResponse.ExtraFields.RawResponse = rawResponse } - return bifrostResponse, nil } @@ -1073,8 +1213,12 @@ func (provider *AnthropicProvider) ResponsesStream(ctx *schemas.BifrostContext, return nil, err } - // Convert to Anthropic format using the centralized converter - jsonBody, err := getRequestBodyForResponses(ctx, request, true, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + jsonBody, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ + Provider: schemas.Anthropic, + IsStreaming: true, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }) if err != nil { return nil, err } @@ -1105,6 +1249,7 @@ func (provider *AnthropicProvider) ResponsesStream(ctx *schemas.BifrostContext, provider.GetProviderKey(), postHookRunner, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -1126,6 +1271,7 @@ func HandleAnthropicResponsesStream( providerName schemas.ModelProvider, postHookRunner schemas.PostHookRunner, postResponseConverter func(*schemas.BifrostResponsesStreamResponse) *schemas.BifrostResponsesStreamResponse, + signer providerUtils.BodySigner, logger schemas.Logger, postHookSpanFinalizer func(context.Context), ) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { @@ -1155,6 +1301,19 @@ func HandleAnthropicResponsesStream( // Set body usedLargePayloadBody := setAnthropicRequestBody(ctx, req, jsonBody) + // Sign the exact body bytes set on the request, when a signer is supplied (e.g. AWS SigV4 + // for Bedrock Mantle). Done after the body is set so the signature covers what is actually sent. + if signer != nil { + sigHeaders, bErr := signer(jsonBody) + if bErr != nil { + defer providerUtils.ReleaseStreamingResponse(ctx, resp) + return nil, bErr + } + for k, v := range sigHeaders { + req.Header.Set(k, v) + } + } + // Close the connection after this streaming response instead of returning it to // the keep-alive pool. fasthttp can otherwise reuse a streaming connection whose // reader is still in a torn state (body not fully drained, or the idle-timeout/ @@ -1176,6 +1335,7 @@ func HandleAnthropicResponsesStream( startTime := time.Now() // Make the request err := activeClient.Do(req, resp) + latency := time.Since(startTime) if usedLargePayloadBody { providerUtils.DrainLargePayloadRemainder(ctx) } @@ -1189,16 +1349,16 @@ func HandleAnthropicResponsesStream( Message: schemas.ErrRequestCancelled, Error: err, }, - }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Request failed before the first response byte (server closed an idle/pooled connection, // broken pipe, connection refused, DNS failure, etc.). Surface as a retriable upstream // connection error (502) so executeRequestWithRetries honors max_retries, matching the // non-streaming path - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Store provider response headers in context before status check so error responses also forward them @@ -1207,7 +1367,7 @@ func HandleAnthropicResponsesStream( // Check for HTTP errors if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) - return nil, providerUtils.EnrichError(ctx, parseAnthropicError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseAnthropicError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Large payload streaming passthrough — pipe raw upstream SSE to client @@ -1265,10 +1425,29 @@ func HandleAnthropicResponsesStream( // Track minimal state needed for response format usage := &schemas.ResponsesResponseUsage{} + billedUsage := &schemas.BifrostLLMUsage{} + // Register the accumulating usage handle so a mid-stream cancel/timeout + // can bill for Anthropic Responses usage already reported by message_start + // or message_delta events before the stream was interrupted. + ctx.SetValue(schemas.BifrostContextKeyStreamAccumulatedUsage, billedUsage) + + usageNormalized := false + normalizeBilledUsage := func() { + if usageNormalized { + return + } + usageNormalized = true + normalizeCachedUsage(billedUsage) + } + defer func() { + if ctx.Err() != nil { + normalizeBilledUsage() + } + }() // Create stream state for stateful conversions - streamState := acquireAnthropicResponsesStreamState() - defer releaseAnthropicResponsesStreamState(streamState) + streamState := AcquireAnthropicResponsesStreamState() + defer ReleaseAnthropicResponsesStreamState(streamState) // Set structured output tool name if present if toolName, ok := ctx.Value(schemas.BifrostContextKeyStructuredOutputToolName).(string); ok { @@ -1320,49 +1499,10 @@ func HandleAnthropicResponsesStream( } if usageToProcess != nil { - // Collect usage information and send at the end of the stream - // Here in some cases usage comes before final message - // So we need to check if the response.Usage is nil and then if usage != nil - // then add up all tokens - if usageToProcess.InputTokens > usage.InputTokens { - usage.InputTokens = usageToProcess.InputTokens - } - if usageToProcess.OutputTokens > usage.OutputTokens { - usage.OutputTokens = usageToProcess.OutputTokens - } - calculatedTotal := usage.InputTokens + usage.OutputTokens - if calculatedTotal > usage.TotalTokens { - usage.TotalTokens = calculatedTotal - } - // Handle cached tokens if present - if usageToProcess.CacheReadInputTokens > 0 { - if usage.InputTokensDetails == nil { - usage.InputTokensDetails = &schemas.ResponsesResponseInputTokens{} - } - if usageToProcess.CacheReadInputTokens > usage.InputTokensDetails.CachedReadTokens { - usage.InputTokensDetails.CachedReadTokens = usageToProcess.CacheReadInputTokens - } - } - // Handle cached tokens if present - if usageToProcess.CacheCreationInputTokens > 0 { - if usage.InputTokensDetails == nil { - usage.InputTokensDetails = &schemas.ResponsesResponseInputTokens{} - } - if usageToProcess.CacheCreationInputTokens > usage.InputTokensDetails.CachedWriteTokens { - usage.InputTokensDetails.CachedWriteTokens = usageToProcess.CacheCreationInputTokens - } - if usageToProcess.CacheCreation.Ephemeral5mInputTokens > 0 || usageToProcess.CacheCreation.Ephemeral1hInputTokens > 0 { - if usage.InputTokensDetails.CachedWriteTokenDetails == nil { - usage.InputTokensDetails.CachedWriteTokenDetails = &schemas.ChatCachedWriteTokenDetails{} - } - if usageToProcess.CacheCreation.Ephemeral5mInputTokens > usage.InputTokensDetails.CachedWriteTokenDetails.CachedWriteTokens5m { - usage.InputTokensDetails.CachedWriteTokenDetails.CachedWriteTokens5m = usageToProcess.CacheCreation.Ephemeral5mInputTokens - } - if usageToProcess.CacheCreation.Ephemeral1hInputTokens > usage.InputTokensDetails.CachedWriteTokenDetails.CachedWriteTokens1h { - usage.InputTokensDetails.CachedWriteTokenDetails.CachedWriteTokens1h = usageToProcess.CacheCreation.Ephemeral1hInputTokens - } - } - } + // Collect usage information and send at the end of the stream. + // Also mirror it into billedUsage so cancellation/timeout paths can + // charge for provider-reported usage before the final chunk arrives. + accumulateAnthropicResponsesUsage(usage, billedUsage, usageToProcess) } responses, bifrostErr, isLastChunk := event.ToBifrostResponsesStream(ctx, chunkIndex, streamState) @@ -1501,24 +1641,24 @@ func (provider *AnthropicProvider) BatchCreate(ctx *schemas.BifrostContext, key providerUtils.DrainLargePayloadRemainder(ctx) } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Handle error response if resp.StatusCode() != fasthttp.StatusOK { provider.logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) - return nil, parseAnthropicError(resp) + return nil, providerUtils.SetErrorLatency(parseAnthropicError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } var anthropicResp AnthropicBatchResponse rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(body, &anthropicResp, jsonData, sendBackRawRequest, sendBackRawResponse) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, sendBackRawRequest, sendBackRawResponse, latency) } return anthropicResp.ToBifrostBatchCreateResponse(latency, sendBackRawRequest, sendBackRawResponse, rawRequest, rawResponse), nil @@ -1596,7 +1736,7 @@ func (provider *AnthropicProvider) BatchList(ctx *schemas.BifrostContext, keys [ // Handle error response if resp.StatusCode() != fasthttp.StatusOK { provider.logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) - return nil, parseAnthropicError(resp) + return nil, providerUtils.SetErrorLatency(parseAnthropicError(resp), latency) } body, decodeErr := providerUtils.CheckAndDecodeBody(resp) @@ -2030,7 +2170,27 @@ func (provider *AnthropicProvider) FileUpload(ctx *schemas.BifrostContext, key s if filename == "" { filename = "file" } - part, err := writer.CreateFormFile("file", filename) + contentType := "" + if request.ContentType != nil { + contentType = strings.TrimSpace(*request.ContentType) + } + if request.ContentType != nil { + ct := strings.TrimSpace(*request.ContentType) + if strings.ContainsAny(ct, "\r\n") { + return nil, providerUtils.NewBifrostOperationError("invalid content type: %s", fmt.Errorf("contains CR or LF characters")) + } + contentType = ct + } + var part io.Writer + var err error + if contentType != "" { + header := make(textproto.MIMEHeader) + header.Set("Content-Disposition", multipart.FileContentDisposition("file", filename)) + header.Set("Content-Type", contentType) + part, err = writer.CreatePart(header) + } else { + part, err = writer.CreateFormFile("file", filename) + } if err != nil { return nil, providerUtils.NewBifrostOperationError("failed to create form file", err) } @@ -2071,7 +2231,7 @@ func (provider *AnthropicProvider) FileUpload(ctx *schemas.BifrostContext, key s // Handle error response if resp.StatusCode() != fasthttp.StatusOK && resp.StatusCode() != fasthttp.StatusCreated { provider.logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) - return nil, parseAnthropicError(resp) + return nil, providerUtils.SetErrorLatency(parseAnthropicError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -2161,7 +2321,7 @@ func (provider *AnthropicProvider) FileList(ctx *schemas.BifrostContext, keys [] // Handle error response if resp.StatusCode() != fasthttp.StatusOK { provider.logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) - return nil, parseAnthropicError(resp) + return nil, providerUtils.SetErrorLatency(parseAnthropicError(resp), latency) } body, decodeErr := providerUtils.CheckAndDecodeBody(resp) @@ -2498,17 +2658,23 @@ func (provider *AnthropicProvider) CountTokens(ctx *schemas.BifrostContext, key if err := providerUtils.CheckOperationAllowed(schemas.Anthropic, provider.customProviderConfig, schemas.CountTokensRequest); err != nil { return nil, err } - jsonBody, err := getRequestBodyForResponses(ctx, request, false, []string{"max_tokens", "temperature"}, provider.sendBackRawRequest, provider.sendBackRawResponse) + jsonBody, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ + Provider: schemas.Anthropic, + IsStreaming: false, + ExcludeFields: []string{"max_tokens", "temperature"}, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }) if err != nil { return nil, err } - responseBody, latency, providerResponseHeaders, bifrostErr := provider.completeRequest(ctx, jsonBody, provider.buildRequestURL(ctx, "/v1/messages/count_tokens", schemas.CountTokensRequest), key.Value.GetValue(), schemas.CountTokensRequest) + responseBody, latency, providerResponseHeaders, bifrostErr := completeRequest(ctx, provider.client, provider.buildRequestURL(ctx, "/v1/messages/count_tokens", schemas.CountTokensRequest), jsonBody, provider.anthropicRequestHeaders(ctx, key), provider.networkConfig.ExtraHeaders, provider.networkConfig.BetaHeaderOverrides, provider.GetProviderKey(), schemas.CountTokensRequest, nil, provider.logger) if providerResponseHeaders != nil { ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } anthropicResponse := &AnthropicCountTokensResponse{} @@ -2521,7 +2687,7 @@ func (provider *AnthropicProvider) CountTokens(ctx *schemas.BifrostContext, key ) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } response := anthropicResponse.ToBifrostCountTokensResponse(request.Model) @@ -2732,22 +2898,24 @@ func (provider *AnthropicProvider) PassthroughStream( fasthttpReq.SetBody(req.Body) activeClient := providerUtils.PrepareResponseStreaming(ctx, provider.streamingClient, resp) - if err := activeClient.Do(fasthttpReq, resp); err != nil { + err := activeClient.Do(fasthttpReq, resp) + latency := time.Since(startTime) + if err != nil { providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } - return nil, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), latency) } headers := providerUtils.ExtractPassthroughProviderResponseHeaders(resp) diff --git a/core/providers/anthropic/cancelbilling_test.go b/core/providers/anthropic/cancelbilling_test.go index 59dffd4db3b..9713986d84a 100644 --- a/core/providers/anthropic/cancelbilling_test.go +++ b/core/providers/anthropic/cancelbilling_test.go @@ -43,3 +43,61 @@ func TestNormalizeCachedUsage_NoCacheDetailsIsNoOp(t *testing.T) { func TestNormalizeCachedUsage_NilSafe(t *testing.T) { normalizeCachedUsage(nil) // must not panic } + +func TestAccumulateAnthropicResponsesUsage_MirrorsCacheIntoBilledUsage(t *testing.T) { + responseUsage := &schemas.ResponsesResponseUsage{} + billedUsage := &schemas.BifrostLLMUsage{} + upstreamUsage := &AnthropicUsage{ + InputTokens: 2, + CacheReadInputTokens: 1300, + CacheCreationInputTokens: 78456, + CacheCreation: AnthropicUsageCacheCreation{ + Ephemeral1hInputTokens: 78456, + }, + OutputTokens: 3, + } + + accumulateAnthropicResponsesUsage(responseUsage, billedUsage, upstreamUsage) + + if responseUsage.InputTokens != 2 || responseUsage.OutputTokens != 3 || responseUsage.TotalTokens != 5 { + t.Fatalf("unexpected response usage before final normalization: %+v", responseUsage) + } + if responseUsage.InputTokensDetails == nil { + t.Fatal("expected response cache details") + } + if responseUsage.InputTokensDetails.CachedReadTokens != 1300 { + t.Fatalf("response cached read = %d, want 1300", responseUsage.InputTokensDetails.CachedReadTokens) + } + if responseUsage.InputTokensDetails.CachedWriteTokens != 78456 { + t.Fatalf("response cached write = %d, want 78456", responseUsage.InputTokensDetails.CachedWriteTokens) + } + if responseUsage.InputTokensDetails.CachedWriteTokenDetails == nil || + responseUsage.InputTokensDetails.CachedWriteTokenDetails.CachedWriteTokens1h != 78456 { + t.Fatalf("response cached write details = %+v, want 1h=78456", responseUsage.InputTokensDetails.CachedWriteTokenDetails) + } + + if billedUsage.PromptTokens != 2 || billedUsage.CompletionTokens != 3 || billedUsage.TotalTokens != 5 { + t.Fatalf("unexpected billed usage before normalization: %+v", billedUsage) + } + if billedUsage.PromptTokensDetails == nil { + t.Fatal("expected billed cache details") + } + if billedUsage.PromptTokensDetails.CachedReadTokens != 1300 { + t.Fatalf("billed cached read = %d, want 1300", billedUsage.PromptTokensDetails.CachedReadTokens) + } + if billedUsage.PromptTokensDetails.CachedWriteTokens != 78456 { + t.Fatalf("billed cached write = %d, want 78456", billedUsage.PromptTokensDetails.CachedWriteTokens) + } + if billedUsage.PromptTokensDetails.CachedWriteTokenDetails == nil || + billedUsage.PromptTokensDetails.CachedWriteTokenDetails.CachedWriteTokens1h != 78456 { + t.Fatalf("billed cached write details = %+v, want 1h=78456", billedUsage.PromptTokensDetails.CachedWriteTokenDetails) + } + + normalizeCachedUsage(billedUsage) + if billedUsage.PromptTokens != 79758 { + t.Fatalf("normalized billed prompt = %d, want 79758", billedUsage.PromptTokens) + } + if billedUsage.TotalTokens != 79761 { + t.Fatalf("normalized billed total = %d, want 79761", billedUsage.TotalTokens) + } +} diff --git a/core/providers/anthropic/chat.go b/core/providers/anthropic/chat.go index 0287287708d..600f5fc8547 100644 --- a/core/providers/anthropic/chat.go +++ b/core/providers/anthropic/chat.go @@ -410,8 +410,9 @@ func ToAnthropicChatRequest(ctx *schemas.BifrostContext, bifrostReq *schemas.Bif anthropicReq.MCPServers = servers } if bifrostReq.Params.ResponseFormat != nil { - // Vertex doesn't support native structured outputs, so convert to tool - if bifrostReq.Provider == schemas.Vertex { + // Vertex and Bedrock Mantle don't accept native structured outputs + // (output_config.format), so convert to a tool instead. + if bifrostReq.Provider == schemas.Vertex || bifrostReq.Provider == schemas.BedrockMantle { responseFormatTool := convertChatResponseFormatToTool(ctx, bifrostReq.Params) if responseFormatTool != nil { anthropicReq.Tools = append(anthropicReq.Tools, *responseFormatTool) @@ -734,6 +735,17 @@ func ToAnthropicChatRequest(ctx *schemas.BifrostContext, bifrostReq *schemas.Bif // First add reasoning details if msg.ChatAssistantMessage != nil && msg.ChatAssistantMessage.ReasoningDetails != nil { for _, reasoningDetail := range msg.ChatAssistantMessage.ReasoningDetails { + // reasoning.encrypted details carrying data hold an anthropic + // redacted_thinking payload; replay the block as-is so the API + // can decrypt it. Encrypted details without data (e.g. gemini + // thought signatures) keep the thinking-block mapping below. + if reasoningDetail.Type == schemas.BifrostReasoningDetailsTypeEncrypted && reasoningDetail.Data != nil && *reasoningDetail.Data != "" { + content = append(content, AnthropicContentBlock{ + Type: AnthropicContentBlockTypeRedactedThinking, + Data: reasoningDetail.Data, + }) + continue + } content = append(content, AnthropicContentBlock{ Type: AnthropicContentBlockTypeThinking, Signature: reasoningDetail.Signature, @@ -924,6 +936,19 @@ func (response *AnthropicMessageResponse) ToBifrostChatResponse(ctx *schemas.Bif if c.Thinking != nil { reasoningText += *c.Thinking + "\n" } + case AnthropicContentBlockTypeRedactedThinking: + // Redacted thinking is an opaque encrypted payload. Preserve it as a + // reasoning.encrypted detail: Anthropic requires thinking and + // redacted_thinking blocks to be replayed unmodified on the next + // turn during tool use, and rejects the request when they are + // dropped from the latest assistant message. + if c.Data != nil && *c.Data != "" { + reasoningDetails = append(reasoningDetails, schemas.ChatReasoningDetails{ + Index: len(reasoningDetails), + Type: schemas.BifrostReasoningDetailsTypeEncrypted, + Data: c.Data, + }) + } } } } @@ -1171,6 +1196,14 @@ type AnthropicStreamState struct { // with an empty accumulated arguments string. We track this so content_block_stop // can flush a synthetic "{}" delta when no real arguments arrived. sawArgsDelta map[int]bool + // reasoningDetailIdxByBlock maps an anthropic content_block index to a + // stable reasoning_details index. Thinking and redacted_thinking blocks + // share one sequence, so mixed reasoning streams keep distinct detail + // entries; the accumulator and replaying clients group reasoning deltas + // by that index, and entries merged across blocks lose their type and + // payload on replay. + reasoningDetailIdxByBlock map[int]int + nextReasoningDetailIdx int } // NewAnthropicStreamState returns an initialised stream state for one streaming response. @@ -1178,7 +1211,20 @@ func NewAnthropicStreamState() *AnthropicStreamState { return &AnthropicStreamState{ contentBlockToToolCallIdx: make(map[int]int), sawArgsDelta: make(map[int]bool), + reasoningDetailIdxByBlock: make(map[int]int), + } +} + +// reasoningDetailIndex returns the stable reasoning_details index for an +// anthropic content block, allocating the next one on first use. +func (state *AnthropicStreamState) reasoningDetailIndex(blockIndex int) int { + if idx, ok := state.reasoningDetailIdxByBlock[blockIndex]; ok { + return idx } + idx := state.nextReasoningDetailIdx + state.reasoningDetailIdxByBlock[blockIndex] = idx + state.nextReasoningDetailIdx++ + return idx } // ToBifrostChatCompletionStream converts an Anthropic stream event to a Bifrost Chat Completion Stream response @@ -1222,45 +1268,78 @@ func (chunk *AnthropicStreamEvent) ToBifrostChatCompletionStream(ctx *schemas.Bi return nil, nil, true case AnthropicStreamEventTypeContentBlockStart: - // Emit tool-call metadata when starting a tool_use content block - if chunk.Index != nil && chunk.ContentBlock != nil && chunk.ContentBlock.Type == AnthropicContentBlockTypeToolUse { - // Check if this is the structured output tool - if so, skip emitting tool call metadata - if structuredOutputToolName != "" && chunk.ContentBlock.Name != nil && *chunk.ContentBlock.Name == structuredOutputToolName { - // Skip emitting tool call for structured output - it will be emitted as content later - return nil, nil, false - } + if chunk.Index != nil && chunk.ContentBlock != nil { + switch chunk.ContentBlock.Type { + case AnthropicContentBlockTypeToolUse: + // Check if this is the structured output tool - if so, skip emitting tool call metadata + if structuredOutputToolName != "" && chunk.ContentBlock.Name != nil && *chunk.ContentBlock.Name == structuredOutputToolName { + // Skip emitting tool call for structured output - it will be emitted as content later + return nil, nil, false + } - // Assign the next sequential tool-call index - toolCallIdx := state.nextToolCallIndex - state.contentBlockToToolCallIdx[*chunk.Index] = toolCallIdx - state.nextToolCallIndex++ + // Assign the next sequential tool-call index + toolCallIdx := state.nextToolCallIndex + state.contentBlockToToolCallIdx[*chunk.Index] = toolCallIdx + state.nextToolCallIndex++ - // Create streaming response with tool call metadata - streamResponse := &schemas.BifrostChatResponse{ - Object: "chat.completion.chunk", - Choices: []schemas.BifrostResponseChoice{ - { - Index: 0, - ChatStreamResponseChoice: &schemas.ChatStreamResponseChoice{ - Delta: &schemas.ChatStreamResponseChoiceDelta{ - ToolCalls: []schemas.ChatAssistantMessageToolCall{ - { - Index: uint16(toolCallIdx), - Type: schemas.Ptr(string(schemas.ChatToolTypeFunction)), - ID: chunk.ContentBlock.ID, - Function: schemas.ChatAssistantMessageToolCallFunction{ - Name: chunk.ContentBlock.Name, - Arguments: "", // Empty arguments initially, will be filled by subsequent deltas + // Create streaming response with tool call metadata + streamResponse := &schemas.BifrostChatResponse{ + Object: "chat.completion.chunk", + Choices: []schemas.BifrostResponseChoice{ + { + Index: 0, + ChatStreamResponseChoice: &schemas.ChatStreamResponseChoice{ + Delta: &schemas.ChatStreamResponseChoiceDelta{ + ToolCalls: []schemas.ChatAssistantMessageToolCall{ + { + Index: uint16(toolCallIdx), + Type: schemas.Ptr(string(schemas.ChatToolTypeFunction)), + ID: chunk.ContentBlock.ID, + Function: schemas.ChatAssistantMessageToolCallFunction{ + Name: chunk.ContentBlock.Name, + Arguments: "", // Empty arguments initially, will be filled by subsequent deltas + }, }, }, }, }, }, }, - }, - } + } - return streamResponse, nil, false + return streamResponse, nil, false + + case AnthropicContentBlockTypeRedactedThinking: + // Redacted thinking blocks arrive complete in content_block_start (no + // deltas follow). Surface the encrypted payload as a reasoning.encrypted + // detail so clients can replay it on the next turn; Anthropic rejects + // tool-use follow-ups whose latest assistant message dropped it. + if chunk.ContentBlock.Data == nil || *chunk.ContentBlock.Data == "" { + return nil, nil, false + } + return &schemas.BifrostChatResponse{ + Object: "chat.completion.chunk", + Choices: []schemas.BifrostResponseChoice{ + { + Index: 0, + ChatStreamResponseChoice: &schemas.ChatStreamResponseChoice{ + Delta: &schemas.ChatStreamResponseChoiceDelta{ + ReasoningDetails: []schemas.ChatReasoningDetails{ + { + Index: state.reasoningDetailIndex(*chunk.Index), + Type: schemas.BifrostReasoningDetailsTypeEncrypted, + Data: chunk.ContentBlock.Data, + }, + }, + }, + }, + }, + }, + }, nil, false + + default: + return nil, nil, false + } } return nil, nil, false @@ -1348,7 +1427,7 @@ func (chunk *AnthropicStreamEvent) ToBifrostChatCompletionStream(ctx *schemas.Bi Reasoning: schemas.Ptr(thinkingText), ReasoningDetails: []schemas.ChatReasoningDetails{ { - Index: 0, + Index: state.reasoningDetailIndex(*chunk.Index), Type: schemas.BifrostReasoningDetailsTypeText, Text: schemas.Ptr(thinkingText), }, @@ -1374,7 +1453,7 @@ func (chunk *AnthropicStreamEvent) ToBifrostChatCompletionStream(ctx *schemas.Bi Delta: &schemas.ChatStreamResponseChoiceDelta{ ReasoningDetails: []schemas.ChatReasoningDetails{ { - Index: 0, + Index: state.reasoningDetailIndex(*chunk.Index), Type: schemas.BifrostReasoningDetailsTypeText, Signature: chunk.Delta.Signature, }, diff --git a/core/providers/anthropic/chat_test.go b/core/providers/anthropic/chat_test.go index 406c70c931d..a189ccc8d5f 100644 --- a/core/providers/anthropic/chat_test.go +++ b/core/providers/anthropic/chat_test.go @@ -6,6 +6,8 @@ import ( "strings" "testing" + "github.com/bytedance/sonic" + "github.com/maximhq/bifrost/core/providers/openai" "github.com/maximhq/bifrost/core/schemas" ) @@ -79,6 +81,64 @@ func TestToAnthropicChatRequest_PreservesPropertyOrder(t *testing.T) { } } +func TestToAnthropicChatRequest_OpenAICompatibleFileIDUsesFileSource(t *testing.T) { + body := `{ + "model": "anthropic/claude-sonnet-4-5-20250929", + "messages": [{ + "role": "user", + "content": [ + {"type": "text", "text": "Read the attached PDF."}, + { + "type": "file", + "file": { + "file_id": "file_abc123", + "filename": "tiny.pdf", + "format": "application/pdf" + } + } + ] + }] + }` + + var openAIReq openai.OpenAIChatRequest + if err := sonic.Unmarshal([]byte(body), &openAIReq); err != nil { + t.Fatalf("unmarshal OpenAI-compatible request: %v", err) + } + + ctx := schemas.NewBifrostContext(nil, schemas.NoDeadline) + bifrostReq := openAIReq.ToBifrostChatRequest(ctx) + result, err := ToAnthropicChatRequest(ctx, bifrostReq) + if err != nil { + t.Fatalf("convert to Anthropic request: %v", err) + } + + if len(result.Messages) != 1 { + t.Fatalf("expected one message, got %d", len(result.Messages)) + } + blocks := result.Messages[0].Content.ContentBlocks + if len(blocks) != 2 { + t.Fatalf("expected two content blocks, got %d", len(blocks)) + } + + documentBlock := blocks[1] + if documentBlock.Type != AnthropicContentBlockTypeDocument { + t.Fatalf("expected document block, got %q", documentBlock.Type) + } + if documentBlock.Title == nil || *documentBlock.Title != "tiny.pdf" { + t.Fatalf("expected document title tiny.pdf, got %v", documentBlock.Title) + } + if documentBlock.Source == nil || documentBlock.Source.SourceObj == nil { + t.Fatalf("expected document source object, got %#v", documentBlock.Source) + } + source := documentBlock.Source.SourceObj + if source.Type != "file" { + t.Fatalf("expected source type file, got %q", source.Type) + } + if source.FileID == nil || *source.FileID != "file_abc123" { + t.Fatalf("expected source file_id file_abc123, got %v", source.FileID) + } +} + func TestToAnthropicChatRequest_CachingDeterminism(t *testing.T) { makeReq := func(props *schemas.OrderedMap) *schemas.BifrostChatRequest { return &schemas.BifrostChatRequest{ diff --git a/core/providers/anthropic/codeexecution_test.go b/core/providers/anthropic/codeexecution_test.go index 09e2a02702f..587159481f3 100644 --- a/core/providers/anthropic/codeexecution_test.go +++ b/core/providers/anthropic/codeexecution_test.go @@ -323,8 +323,8 @@ func TestCodeExecution_BashStream(t *testing.T) { // -> ctx bridge (mirrored from anthropic.go) lets the reverse converter skip a // duplicate message_delta on response.completed. ctx.SetValue(schemas.BifrostContextKeyIntegrationType, "anthropic") - state := acquireAnthropicResponsesStreamState() - defer releaseAnthropicResponsesStreamState(state) + state := AcquireAnthropicResponsesStreamState() + defer ReleaseAnthropicResponsesStreamState(state) var emitted []*schemas.BifrostResponsesStreamResponse seq := 0 @@ -519,8 +519,8 @@ var textEditorCodeExecStreamEvents = []string{ func TestCodeExecution_TextEditorStream(t *testing.T) { ctx := schemas.NewBifrostContext(nil, time.Time{}) ctx.SetValue(schemas.BifrostContextKeyIntegrationType, "anthropic") - state := acquireAnthropicResponsesStreamState() - defer releaseAnthropicResponsesStreamState(state) + state := AcquireAnthropicResponsesStreamState() + defer ReleaseAnthropicResponsesStreamState(state) var emitted []*schemas.BifrostResponsesStreamResponse seq := 0 @@ -791,8 +791,8 @@ func TestGenerateSyntheticInputJSONDeltas_UTF8(t *testing.T) { func TestCodeExecution_ProgrammaticStreamRoundTrip(t *testing.T) { ctx := schemas.NewBifrostContext(nil, time.Time{}) ctx.SetValue(schemas.BifrostContextKeyIntegrationType, "anthropic") - state := acquireAnthropicResponsesStreamState() - defer releaseAnthropicResponsesStreamState(state) + state := AcquireAnthropicResponsesStreamState() + defer ReleaseAnthropicResponsesStreamState(state) var back []*AnthropicStreamEvent seq := 0 diff --git a/core/providers/anthropic/emptytoolresult_test.go b/core/providers/anthropic/emptytoolresult_test.go new file mode 100644 index 00000000000..99137a7f20a --- /dev/null +++ b/core/providers/anthropic/emptytoolresult_test.go @@ -0,0 +1,61 @@ +package anthropic + +import ( + "testing" + + "github.com/maximhq/bifrost/core/schemas" +) + +// TestConvertToolResultWithEmptyContent verifies that an Anthropic tool_result +// block with an empty content array converts to a function_call_output whose +// Output serializes cleanly. An all-nil output struct fails MarshalJSON and +// poisons every enclosing structure (conversation histories, log rows). +func TestConvertToolResultWithEmptyContent(t *testing.T) { + role := schemas.ResponsesInputMessageRoleUser + blocks := []AnthropicContentBlock{ + { + Type: AnthropicContentBlockTypeToolResult, + ToolUseID: schemas.Ptr("toolu_empty"), + Content: &AnthropicContent{ContentBlocks: []AnthropicContentBlock{}}, + }, + } + + msgs := convertAnthropicContentBlocksToResponsesMessagesGrouped(blocks, &role, false) + if len(msgs) != 1 { + t.Fatalf("expected 1 converted message, got %d", len(msgs)) + } + out := msgs[0].ResponsesToolMessage.Output + if out == nil { + t.Fatal("expected non-nil tool message output") + } + if _, err := schemas.MarshalSorted(out); err != nil { + t.Fatalf("converted empty tool_result output must marshal, got: %v", err) + } + if _, err := schemas.MarshalSorted(msgs); err != nil { + t.Fatalf("converted messages must marshal as a slice, got: %v", err) + } +} + +// TestConvertToolResultWithUnsupportedBlocks verifies tool_result content made +// solely of block types the converter does not map (e.g. document) still +// yields a serializable output. +func TestConvertToolResultWithUnsupportedBlocks(t *testing.T) { + role := schemas.ResponsesInputMessageRoleUser + blocks := []AnthropicContentBlock{ + { + Type: AnthropicContentBlockTypeToolResult, + ToolUseID: schemas.Ptr("toolu_doc"), + Content: &AnthropicContent{ContentBlocks: []AnthropicContentBlock{ + {Type: AnthropicContentBlockTypeDocument}, + }}, + }, + } + + msgs := convertAnthropicContentBlocksToResponsesMessagesGrouped(blocks, &role, false) + if len(msgs) != 1 { + t.Fatalf("expected 1 converted message, got %d", len(msgs)) + } + if _, err := schemas.MarshalSorted(msgs); err != nil { + t.Fatalf("converted messages must marshal, got: %v", err) + } +} diff --git a/core/providers/anthropic/passthroughstream_test.go b/core/providers/anthropic/passthroughstream_test.go new file mode 100644 index 00000000000..e523ede75ff --- /dev/null +++ b/core/providers/anthropic/passthroughstream_test.go @@ -0,0 +1,597 @@ +package anthropic + +import ( + "fmt" + "testing" + "time" + + "github.com/bytedance/sonic" + "github.com/maximhq/bifrost/core/schemas" + "github.com/tidwall/gjson" +) + +// This file reproduces and guards the Claude Code advisor/server-tool streaming +// bug: on the Anthropic passthrough path (gated by IsClaudeCodeRequest), raw +// upstream frames are forwarded verbatim (upstream content-block indices) while +// the converter re-synthesizes server-tool result blocks (its own re-numbered +// indices). Mixing the two schemes made a strict client (Claude Code) see a +// content_block_stop/_delta for an index that no content_block_start opened, +// failing the turn with "API Error: Content block not found". +// +// The fix lives in the transport (transports/bifrost-http/integrations/anthropic.go +// mustConvertInPassthrough): server-tool frames and every output_item.added are +// rendered by the converter instead of forwarded raw, keeping the converter's +// block-index allocation authoritative and in lockstep with the surrounding raw +// frames. The passthroughMustConvert helper below mirrors that rule; the transport +// package's TestMustConvertInPassthrough pins the real function against it. + +// --- upstream Anthropic SSE fixture builders (indices match a real stream) --- + +func ptMsgStart() string { + return `{"type":"message_start","message":{"model":"claude-sonnet-4-6","id":"msg_019M","type":"message","role":"assistant","content":[],"stop_reason":null,"usage":{"input_tokens":999,"output_tokens":35}}}` +} +func ptMsgEnd() []string { + return []string{ + `{"type":"message_delta","delta":{"stop_reason":"end_turn"},"usage":{"output_tokens":3777}}`, + `{"type":"message_stop"}`, + } +} +func ptThinking(i int) []string { + return []string{ + fmt.Sprintf(`{"type":"content_block_start","index":%d,"content_block":{"type":"thinking","thinking":""}}`, i), + fmt.Sprintf(`{"type":"content_block_delta","index":%d,"delta":{"type":"thinking_delta","thinking":"hmm"}}`, i), + fmt.Sprintf(`{"type":"content_block_delta","index":%d,"delta":{"type":"signature_delta","signature":"abc"}}`, i), + fmt.Sprintf(`{"type":"content_block_stop","index":%d}`, i), + } +} + +// ptAdvisor emits a server_tool_use(advisor) at index i and its advisor_tool_result +// at index i+1 — the shape captured from a real claude -p --advisor turn. +func ptAdvisor(i int, id string) []string { + return []string{ + fmt.Sprintf(`{"type":"content_block_start","index":%d,"content_block":{"type":"server_tool_use","id":%q,"name":"advisor","input":{}}}`, i, id), + fmt.Sprintf(`{"type":"content_block_delta","index":%d,"delta":{"type":"input_json_delta","partial_json":""}}`, i), + fmt.Sprintf(`{"type":"content_block_stop","index":%d}`, i), + fmt.Sprintf(`{"type":"content_block_start","index":%d,"content_block":{"type":"advisor_tool_result","tool_use_id":%q,"content":{"type":"advisor_result","text":"graceful means X."}}}`, i+1, id), + fmt.Sprintf(`{"type":"content_block_stop","index":%d}`, i+1), + } +} +func ptWebSearch(i int, id string) []string { + return []string{ + fmt.Sprintf(`{"type":"content_block_start","index":%d,"content_block":{"type":"server_tool_use","id":%q,"name":"web_search","input":{"query":"cats"}}}`, i, id), + fmt.Sprintf(`{"type":"content_block_stop","index":%d}`, i), + fmt.Sprintf(`{"type":"content_block_start","index":%d,"content_block":{"type":"web_search_tool_result","tool_use_id":%q,"content":[{"type":"web_search_result","url":"https://x.com","title":"X","encrypted_content":"e"}]}}`, i+1, id), + fmt.Sprintf(`{"type":"content_block_stop","index":%d}`, i+1), + } +} + +// ptWebSearchNoResults is a web_search whose result block carries no sources +// (empty results / error case). Upstream still emits the web_search_tool_result +// block, consuming an index — so the reverse converter must still advance its +// block counter even though it emits no result content. +func ptWebSearchNoResults(i int, id string) []string { + return []string{ + fmt.Sprintf(`{"type":"content_block_start","index":%d,"content_block":{"type":"server_tool_use","id":%q,"name":"web_search","input":{"query":"cats"}}}`, i, id), + fmt.Sprintf(`{"type":"content_block_stop","index":%d}`, i), + fmt.Sprintf(`{"type":"content_block_start","index":%d,"content_block":{"type":"web_search_tool_result","tool_use_id":%q,"content":[]}}`, i+1, id), + fmt.Sprintf(`{"type":"content_block_stop","index":%d}`, i+1), + } +} +func ptWebFetch(i int, id string) []string { + return []string{ + fmt.Sprintf(`{"type":"content_block_start","index":%d,"content_block":{"type":"server_tool_use","id":%q,"name":"web_fetch","input":{"url":"https://x.com"}}}`, i, id), + fmt.Sprintf(`{"type":"content_block_delta","index":%d,"delta":{"type":"input_json_delta","partial_json":""}}`, i), + fmt.Sprintf(`{"type":"content_block_stop","index":%d}`, i), + fmt.Sprintf(`{"type":"content_block_start","index":%d,"content_block":{"type":"web_fetch_tool_result","tool_use_id":%q,"content":{"type":"web_fetch_result","url":"https://x.com","content":{"type":"text","text":"hi"}}}}`, i+1, id), + fmt.Sprintf(`{"type":"content_block_stop","index":%d}`, i+1), + } +} +func ptCodeExec(i int, id string) []string { + return []string{ + fmt.Sprintf(`{"type":"content_block_start","index":%d,"content_block":{"type":"server_tool_use","id":%q,"name":"bash_code_execution","input":{}}}`, i, id), + fmt.Sprintf(`{"type":"content_block_delta","index":%d,"delta":{"type":"input_json_delta","partial_json":"{\"command\":\"ls\"}"}}`, i), + fmt.Sprintf(`{"type":"content_block_stop","index":%d}`, i), + fmt.Sprintf(`{"type":"content_block_start","index":%d,"content_block":{"type":"bash_code_execution_tool_result","tool_use_id":%q,"content":{"type":"bash_code_execution_result","stdout":"out","stderr":"","return_code":0}}}`, i+1, id), + fmt.Sprintf(`{"type":"content_block_stop","index":%d}`, i+1), + } +} +func ptFuncTool(i int, id, name string) []string { + return []string{ + fmt.Sprintf(`{"type":"content_block_start","index":%d,"content_block":{"type":"tool_use","id":%q,"name":%q,"input":{}}}`, i, id, name), + fmt.Sprintf(`{"type":"content_block_delta","index":%d,"delta":{"type":"input_json_delta","partial_json":"{\"x\":1}"}}`, i), + fmt.Sprintf(`{"type":"content_block_stop","index":%d}`, i), + } +} +func ptComputer(i int, id string) []string { + return []string{ + fmt.Sprintf(`{"type":"content_block_start","index":%d,"content_block":{"type":"tool_use","id":%q,"name":"computer","input":{}}}`, i, id), + fmt.Sprintf(`{"type":"content_block_delta","index":%d,"delta":{"type":"input_json_delta","partial_json":"{\"action\":\"screenshot\"}"}}`, i), + fmt.Sprintf(`{"type":"content_block_stop","index":%d}`, i), + } +} +func ptText(i int) []string { + return []string{ + fmt.Sprintf(`{"type":"content_block_start","index":%d,"content_block":{"type":"text","text":""}}`, i), + fmt.Sprintf(`{"type":"content_block_delta","index":%d,"delta":{"type":"text_delta","text":"OK"}}`, i), + fmt.Sprintf(`{"type":"content_block_stop","index":%d}`, i), + } +} +func ptConcat(parts ...[]string) []string { + var out []string + for _, p := range parts { + out = append(out, p...) + } + return out +} + +// passthroughMustConvert mirrors transports/bifrost-http/integrations/anthropic.go +// mustConvertInPassthrough. Kept in sync by that package's TestMustConvertInPassthrough. +func passthroughMustConvert(r *schemas.BifrostResponsesStreamResponse) bool { + switch r.Type { + case schemas.ResponsesStreamResponseTypeOutputItemAdded: + return true + case schemas.ResponsesStreamResponseTypeOutputItemDone: + if r.Item == nil || r.Item.Type == nil { + return false + } + switch *r.Item.Type { + case schemas.ResponsesMessageTypeAdvisorCall, + schemas.ResponsesMessageTypeWebSearchCall, + schemas.ResponsesMessageTypeWebFetchCall, + schemas.ResponsesMessageTypeCodeInterpreterCall: + return true + } + return false + case schemas.ResponsesStreamResponseTypeWebSearchCallInProgress, + schemas.ResponsesStreamResponseTypeWebSearchCallSearching, + schemas.ResponsesStreamResponseTypeWebSearchCallCompleted, + schemas.ResponsesStreamResponseTypeWebSearchCallResultsAdded, + schemas.ResponsesStreamResponseTypeWebSearchCallResultsCompleted, + schemas.ResponsesStreamResponseTypeWebFetchCallInProgress, + schemas.ResponsesStreamResponseTypeWebFetchCallFetching, + schemas.ResponsesStreamResponseTypeWebFetchCallCompleted, + schemas.ResponsesStreamResponseTypeCodeInterpreterCallInProgress, + schemas.ResponsesStreamResponseTypeCodeInterpreterCallInterpreting, + schemas.ResponsesStreamResponseTypeCodeInterpreterCallCompleted, + schemas.ResponsesStreamResponseTypeCodeInterpreterCallCodeDelta, + schemas.ResponsesStreamResponseTypeCodeInterpreterCallCodeDone: + return true + } + return false +} + +type ptFrame struct { + typ string + idx int + via string +} + +// runAnthropicPassthrough drives the real pipeline: each upstream Anthropic SSE +// frame is converted to bifrost stream responses (ToBifrostResponsesStream); the +// raw upstream frame is attached to exactly one response (response.created, else +// the last — mirroring core/providers/anthropic/anthropic.go's stream loop); then +// the transport passthrough decision picks raw vs converter for each response. +// When applyFix is false it models the pre-fix transport (raw whenever present, +// except ContentPartAdded) to prove the bug reproduces. +func runAnthropicPassthrough(t *testing.T, raws []string, applyFix bool) ([]ptFrame, *schemas.BifrostContext) { + t.Helper() + ctx := schemas.NewBifrostContext(nil, time.Time{}) + // This harness always models the passthrough path (raw frames interleaved), so + // mark the reverse converter accordingly — mirroring the transport, which calls + // SetResponsesStreamPassthrough when shouldUsePassthrough is true. This drives + // the server-tool result-block index bumps that keep converted frames in + // lockstep with the raw ones; the all-normalized path (TestAnthropicConverterOnly_*) + // leaves it unset so indices stay contiguous. + SetResponsesStreamPassthrough(ctx) + state := newAdvisorStreamState() + var out []ptFrame + seq := 0 + for _, raw := range raws { + var chunk AnthropicStreamEvent + if err := sonic.Unmarshal([]byte(raw), &chunk); err != nil { + t.Fatalf("unmarshal event: %v", err) + } + responses, bErr, _ := chunk.ToBifrostResponsesStream(ctx, seq, state) + if bErr != nil { + t.Fatalf("ToBifrostResponsesStream error: %v", bErr) + } + rawIdx := len(responses) - 1 + for j, r := range responses { + if r != nil && r.Type == schemas.ResponsesStreamResponseTypeCreated { + rawIdx = j + break + } + } + for i, r := range responses { + seq++ + useRaw := i == rawIdx && r.Type != schemas.ResponsesStreamResponseTypeContentPartAdded + if applyFix && passthroughMustConvert(r) { + useRaw = false + } + if useRaw { + typ := gjson.Get(raw, "type").String() + if typ == "" { + continue + } + idx := -1 + if g := gjson.Get(raw, "index"); g.Exists() { + idx = int(g.Int()) + } + out = append(out, ptFrame{typ: typ, idx: idx, via: "raw"}) + continue + } + for _, e := range ToAnthropicResponsesStreamResponse(ctx, r) { + idx := -1 + if e.Index != nil { + idx = *e.Index + } + out = append(out, ptFrame{typ: string(e.Type), idx: idx, via: "conv"}) + } + } + } + return out, ctx +} + +// blockFramingProblems models a strict SSE consumer (Claude Code): every +// content_block_delta / content_block_stop must reference an index opened by a +// prior content_block_start, no index may be opened twice, every opened block must +// be closed. +func blockFramingProblems(t *testing.T, frames []ptFrame, dump bool) []string { + t.Helper() + var problems []string + open := map[int]bool{} + for i, f := range frames { + if dump { + t.Logf(" [%02d] %-24s index=%-3d (%s)", i, f.typ, f.idx, f.via) + } + switch f.typ { + case "content_block_start": + if open[f.idx] { + problems = append(problems, fmt.Sprintf("double content_block_start for index %d", f.idx)) + } + open[f.idx] = true + case "content_block_delta": + if !open[f.idx] { + problems = append(problems, fmt.Sprintf("content_block_delta for unopened index %d", f.idx)) + } + case "content_block_stop": + if !open[f.idx] { + problems = append(problems, fmt.Sprintf("content_block_stop for unopened index %d", f.idx)) + } + delete(open, f.idx) + } + } + for idx := range open { + problems = append(problems, fmt.Sprintf("index %d opened but never stopped", idx)) + } + return problems +} + +func TestAnthropicPassthrough_ServerToolIndexConsistency(t *testing.T) { + scenarios := map[string][]string{ + "thinking+advisor+text": ptConcat([]string{ptMsgStart()}, ptThinking(0), ptAdvisor(1, "srv_A"), ptText(3), ptMsgEnd()), + "advisor_first": ptConcat([]string{ptMsgStart()}, ptAdvisor(0, "srv_A"), ptText(2), ptMsgEnd()), + "two_advisors": ptConcat([]string{ptMsgStart()}, ptThinking(0), ptAdvisor(1, "srv_A"), ptAdvisor(3, "srv_B"), ptText(5), ptMsgEnd()), + "websearch": ptConcat([]string{ptMsgStart()}, ptThinking(0), ptWebSearch(1, "srv_W"), ptText(3), ptMsgEnd()), + "websearch_no_results": ptConcat([]string{ptMsgStart()}, ptThinking(0), ptWebSearchNoResults(1, "srv_W"), ptText(3), ptMsgEnd()), + "two_websearch": ptConcat([]string{ptMsgStart()}, ptWebSearch(0, "srv_W"), ptWebSearch(2, "srv_W2"), ptText(4), ptMsgEnd()), + "web_fetch": ptConcat([]string{ptMsgStart()}, ptThinking(0), ptWebFetch(1, "srv_F"), ptText(3), ptMsgEnd()), + "advisor_then_web_fetch": ptConcat([]string{ptMsgStart()}, ptAdvisor(0, "srv_A"), ptWebFetch(2, "srv_F"), ptText(4), ptMsgEnd()), + "two_web_fetch": ptConcat([]string{ptMsgStart()}, ptWebFetch(0, "srv_F"), ptWebFetch(2, "srv_F2"), ptText(4), ptMsgEnd()), + "code_execution": ptConcat([]string{ptMsgStart()}, ptThinking(0), ptCodeExec(1, "srv_C"), ptText(3), ptMsgEnd()), + "func_then_advisor": ptConcat([]string{ptMsgStart()}, ptFuncTool(0, "toolu_1", "Read"), ptAdvisor(1, "srv_A"), ptText(3), ptMsgEnd()), + "advisor_then_func": ptConcat([]string{ptMsgStart()}, ptAdvisor(0, "srv_A"), ptFuncTool(2, "toolu_1", "Read"), ptText(3), ptMsgEnd()), + "advisor_then_websearch": ptConcat([]string{ptMsgStart()}, ptAdvisor(0, "srv_A"), ptWebSearch(2, "srv_W"), ptText(4), ptMsgEnd()), + "computer": ptConcat([]string{ptMsgStart()}, ptThinking(0), ptComputer(1, "toolu_C"), ptText(2), ptMsgEnd()), + "plain_text": ptConcat([]string{ptMsgStart()}, ptText(0), ptMsgEnd()), + } + for name, raws := range scenarios { + t.Run(name, func(t *testing.T) { + frames, ctx := runAnthropicPassthrough(t, raws, true) + if fixed := blockFramingProblems(t, frames, true); len(fixed) > 0 { + t.Errorf("passthrough stream is inconsistent after fix: %v", fixed) + } + // The fix's mechanism: on the passthrough path every output_item.added is + // routed through the converter so allocBlockIndex always runs — so + // blockIndexFor must never miss (a miss = a stop/delta for a block whose + // start was never registered). + if misses := getOrCreateAnthropicToResponsesStreamState(ctx).blockIndexMisses; len(misses) > 0 { + t.Errorf("reverse converter recorded block-index misses on the passthrough path: %v", misses) + } + }) + } +} + +// TestAnthropicPassthrough_ReproducesBugWithoutFix asserts the pre-fix transport +// behavior (forward raw whenever present) produces exactly the doc's failure: a +// content_block_stop for an index the client never opened. This guards the fix +// from silently regressing to a test that passes vacuously. +func TestAnthropicPassthrough_ReproducesBugWithoutFix(t *testing.T) { + stream := ptConcat([]string{ptMsgStart()}, ptThinking(0), ptAdvisor(1, "srv_A"), ptText(3), ptMsgEnd()) + + buggyFrames, _ := runAnthropicPassthrough(t, stream, false) + buggy := blockFramingProblems(t, buggyFrames, false) + if len(buggy) == 0 { + t.Fatal("expected the pre-fix passthrough to produce an inconsistent stream, but it was clean") + } + foundStopUnopened := false + for _, p := range buggy { + if p == "content_block_stop for unopened index 2" { + foundStopUnopened = true + } + } + if !foundStopUnopened { + t.Errorf("expected the documented failure (stop for unopened index 2); got %v", buggy) + } + + // The same stream must be clean once the fix routes server-tool frames through + // the converter. + fixedFrames, _ := runAnthropicPassthrough(t, stream, true) + if fixed := blockFramingProblems(t, fixedFrames, false); len(fixed) > 0 { + t.Errorf("fix did not resolve the inconsistency: %v", fixed) + } +} + +// TestAnthropicConverterOnlyStream_WellFormed covers the all-frames-through-the- +// converter path — i.e. non-Claude-Code clients (plain curl / OpenAI-format), which +// never take the verbatim-passthrough route. It asserts two things over a rich +// server-tool stream: (1) the emitted Anthropic SSE satisfies the content_block +// open/close invariant (every stop/delta references an opened start; nothing left +// open), and (2) blockIndexFor never misses (blockIndexMisses stays empty), i.e. the +// reverse converter's allocator stays in lockstep with itself when it drives every +// frame. +func TestAnthropicConverterOnlyStream_WellFormed(t *testing.T) { + stream := ptConcat([]string{ptMsgStart()}, ptThinking(0), ptAdvisor(1, "srv_A"), ptWebSearch(3, "srv_W"), ptWebFetch(5, "srv_F"), ptText(7), ptMsgEnd()) + ctx := schemas.NewBifrostContext(nil, time.Time{}) + state := newAdvisorStreamState() + var frames []ptFrame + seq := 0 + for _, raw := range stream { + var chunk AnthropicStreamEvent + if err := sonic.Unmarshal([]byte(raw), &chunk); err != nil { + t.Fatalf("unmarshal event: %v", err) + } + responses, bErr, _ := chunk.ToBifrostResponsesStream(ctx, seq, state) + if bErr != nil { + t.Fatalf("ToBifrostResponsesStream error: %v", bErr) + } + for _, r := range responses { + seq++ + for _, e := range ToAnthropicResponsesStreamResponse(ctx, r) { + idx := -1 + if e.Index != nil { + idx = *e.Index + } + frames = append(frames, ptFrame{typ: string(e.Type), idx: idx, via: "conv"}) + } + } + } + if problems := blockFramingProblems(t, frames, false); len(problems) > 0 { + t.Errorf("converter-only stream is not well-formed: %v", problems) + } + rs := getOrCreateAnthropicToResponsesStreamState(ctx) + if len(rs.blockIndexMisses) > 0 { + t.Errorf("reverse converter recorded block-index misses (stop/delta for unregistered blocks): %v", rs.blockIndexMisses) + } +} + +// TestAnthropicConverterOnly_IndicesContiguous guards the all-normalized path +// (OpenAI-via-Anthropic, curl — never Claude Code): with no interleaved raw frames, +// SetResponsesStreamPassthrough is NOT called, so the converter must emit contiguous +// content_block indices 0,1,2,… A gap (an index consumed for a hidden server-tool +// result block but no block emitted) is what a strict Anthropic SDK client sees as a +// missing block; native Anthropic never produces one. Zero-result web_search is the +// case that still collapses a result block — anti-vacuous: before the passthrough +// gating it emitted [0,1,3]. +func TestAnthropicConverterOnly_IndicesContiguous(t *testing.T) { + scenarios := map[string][]string{ + "web_fetch": ptConcat([]string{ptMsgStart()}, ptThinking(0), ptWebFetch(1, "srv_F"), ptText(3), ptMsgEnd()), + "websearch_no_results": ptConcat([]string{ptMsgStart()}, ptThinking(0), ptWebSearchNoResults(1, "srv_W"), ptText(3), ptMsgEnd()), + "websearch_with_results": ptConcat([]string{ptMsgStart()}, ptThinking(0), ptWebSearch(1, "srv_W"), ptText(3), ptMsgEnd()), + "two_web_fetch": ptConcat([]string{ptMsgStart()}, ptWebFetch(0, "srv_F"), ptWebFetch(2, "srv_F2"), ptText(4), ptMsgEnd()), + "web_fetch_then_search": ptConcat([]string{ptMsgStart()}, ptWebFetch(0, "srv_F"), ptWebSearchNoResults(2, "srv_W"), ptText(4), ptMsgEnd()), + } + for name, raws := range scenarios { + t.Run(name, func(t *testing.T) { + // No SetResponsesStreamPassthrough: this is the all-normalized path. + ctx := schemas.NewBifrostContext(nil, time.Time{}) + state := newAdvisorStreamState() + seq := 0 + var starts []int + for _, raw := range raws { + var chunk AnthropicStreamEvent + if err := sonic.Unmarshal([]byte(raw), &chunk); err != nil { + t.Fatalf("unmarshal event: %v", err) + } + responses, bErr, _ := chunk.ToBifrostResponsesStream(ctx, seq, state) + if bErr != nil { + t.Fatalf("ToBifrostResponsesStream error: %v", bErr) + } + for _, r := range responses { + seq++ + for _, e := range ToAnthropicResponsesStreamResponse(ctx, r) { + if e.Type == AnthropicStreamEventTypeContentBlockStart && e.Index != nil { + starts = append(starts, *e.Index) + } + } + } + } + for i, idx := range starts { + if idx != i { + t.Errorf("non-contiguous content_block_start indices %v (position %d is index %d, not %d); native Anthropic never skips an index", starts, i, idx, i) + break + } + } + }) + } +} + +func TestAnthropicWebFetchResultRoundTrip(t *testing.T) { + toolID := "srvtoolu_fetch" + result := &AnthropicContentBlock{ + Type: AnthropicContentBlockTypeWebFetchToolResult, + ToolUseID: &toolID, + Content: &AnthropicContent{ContentObj: &AnthropicContentBlock{ + Type: "web_fetch_result", + URL: schemas.Ptr("https://example.com/article"), + RetrievedAt: schemas.Ptr("2026-07-06T09:00:28Z"), + Content: &AnthropicContent{ContentObj: &AnthropicContentBlock{ + Type: AnthropicContentBlockTypeDocument, + Title: schemas.Ptr("Example Article"), + Source: &AnthropicBlockSource{SourceObj: &AnthropicSource{ + Type: "text", + MediaType: schemas.Ptr("text/plain"), + Data: schemas.Ptr("Full text content"), + }}, + Citations: &AnthropicCitations{Config: &schemas.Citations{Enabled: schemas.Ptr(true)}}, + }}, + }}, + } + + carry := convertAnthropicWebFetchResultToBifrost(result) + if carry == nil || carry.Document == nil || carry.Document.Source == nil { + t.Fatalf("expected typed web_fetch result payload, got %#v", carry) + } + if got := *carry.Document.Source.Data; got != "Full text content" { + t.Fatalf("unexpected carried document data: %q", got) + } + + msg := &schemas.ResponsesMessage{ + ID: &toolID, + Type: schemas.Ptr(schemas.ResponsesMessageTypeWebFetchCall), + Status: schemas.Ptr("completed"), + ResponsesToolMessage: &schemas.ResponsesToolMessage{ + CallID: &toolID, + Action: &schemas.ResponsesToolMessageActionStruct{ + ResponsesWebFetchToolCallAction: &schemas.ResponsesWebFetchToolCallAction{ + Type: "fetch", + URL: "https://example.com/article", + }, + }, + ResponsesWebFetchCall: carry, + }, + } + + blocks := convertBifrostWebFetchCallToAnthropicBlocks(msg) + if len(blocks) != 2 { + t.Fatalf("expected server_tool_use + web_fetch_tool_result, got %d blocks: %#v", len(blocks), blocks) + } + if blocks[1].Type != AnthropicContentBlockTypeWebFetchToolResult { + t.Fatalf("expected web_fetch_tool_result, got %q", blocks[1].Type) + } + inner := getAnthropicContentObject(blocks[1].Content) + if inner == nil || inner.URL == nil || *inner.URL != "https://example.com/article" { + t.Fatalf("expected rebuilt fetch result URL, got %#v", inner) + } + doc := getAnthropicContentObject(inner.Content) + if doc == nil || doc.Source == nil || doc.Source.SourceObj == nil || doc.Source.SourceObj.Data == nil { + t.Fatalf("expected rebuilt document source, got %#v", doc) + } + if got := *doc.Source.SourceObj.Data; got != "Full text content" { + t.Fatalf("unexpected rebuilt document data: %q", got) + } +} + +func TestAnthropicWebFetchTextResultRoundTrip(t *testing.T) { + toolID := "srvtoolu_fetch_text" + result := &AnthropicContentBlock{ + Type: AnthropicContentBlockTypeWebFetchToolResult, + ToolUseID: &toolID, + Content: &AnthropicContent{ContentObj: &AnthropicContentBlock{ + Type: "web_fetch_result", + URL: schemas.Ptr("https://example.com/plain"), + Content: &AnthropicContent{ContentObj: &AnthropicContentBlock{ + Type: AnthropicContentBlockTypeText, + Text: schemas.Ptr("plain fetched text"), + }}, + }}, + } + + carry := convertAnthropicWebFetchResultToBifrost(result) + if carry == nil || carry.Document == nil || carry.Document.Text == nil { + t.Fatalf("expected typed text web_fetch result payload, got %#v", carry) + } + if got := *carry.Document.Text; got != "plain fetched text" { + t.Fatalf("unexpected carried text: %q", got) + } + + msg := &schemas.ResponsesMessage{ + ID: &toolID, + Type: schemas.Ptr(schemas.ResponsesMessageTypeWebFetchCall), + Status: schemas.Ptr("completed"), + ResponsesToolMessage: &schemas.ResponsesToolMessage{ + CallID: &toolID, + ResponsesWebFetchCall: carry, + }, + } + + blocks := convertBifrostWebFetchCallToAnthropicBlocks(msg) + if len(blocks) != 2 { + t.Fatalf("expected server_tool_use + web_fetch_tool_result, got %d blocks: %#v", len(blocks), blocks) + } + inner := getAnthropicContentObject(blocks[1].Content) + if inner == nil || inner.Content == nil { + t.Fatalf("expected rebuilt fetch result content, got %#v", inner) + } + textBlock := getAnthropicContentObject(inner.Content) + if textBlock == nil || textBlock.Text == nil || *textBlock.Text != "plain fetched text" { + t.Fatalf("expected rebuilt text content, got %#v", textBlock) + } +} + +func TestAnthropicWebFetchPassthroughNoResultConsumesHiddenIndex(t *testing.T) { + ctx := schemas.NewBifrostContext(nil, time.Time{}) + SetResponsesStreamPassthrough(ctx) + + toolID := "srvtoolu_fetch_missing_result" + textID := "msg_after_fetch" + addFetch := &schemas.BifrostResponsesStreamResponse{ + Type: schemas.ResponsesStreamResponseTypeOutputItemAdded, + OutputIndex: schemas.Ptr(0), + Item: &schemas.ResponsesMessage{ + ID: &toolID, + Type: schemas.Ptr(schemas.ResponsesMessageTypeWebFetchCall), + Status: schemas.Ptr("in_progress"), + ResponsesToolMessage: &schemas.ResponsesToolMessage{ + CallID: &toolID, + Action: &schemas.ResponsesToolMessageActionStruct{ + ResponsesWebFetchToolCallAction: &schemas.ResponsesWebFetchToolCallAction{ + Type: "fetch", + URL: "https://example.com", + }, + }, + }, + }, + } + doneFetch := &schemas.BifrostResponsesStreamResponse{ + Type: schemas.ResponsesStreamResponseTypeOutputItemDone, + OutputIndex: schemas.Ptr(0), + Item: &schemas.ResponsesMessage{ + ID: &toolID, + Type: schemas.Ptr(schemas.ResponsesMessageTypeWebFetchCall), + Status: schemas.Ptr("completed"), + ResponsesToolMessage: &schemas.ResponsesToolMessage{ + CallID: &toolID, + }, + }, + } + addText := &schemas.BifrostResponsesStreamResponse{ + Type: schemas.ResponsesStreamResponseTypeOutputItemAdded, + OutputIndex: schemas.Ptr(1), + Item: &schemas.ResponsesMessage{ + ID: &textID, + Type: schemas.Ptr(schemas.ResponsesMessageTypeMessage), + Content: &schemas.ResponsesMessageContent{ContentStr: schemas.Ptr("after")}, + }, + } + + _ = ToAnthropicResponsesStreamResponse(ctx, addFetch) + _ = ToAnthropicResponsesStreamResponse(ctx, doneFetch) + events := ToAnthropicResponsesStreamResponse(ctx, addText) + if len(events) == 0 || events[0].Index == nil { + t.Fatalf("expected text start event with index, got %#v", events) + } + if got := *events[0].Index; got != 2 { + t.Fatalf("expected hidden web_fetch result index to be consumed before next block; got index %d", got) + } +} diff --git a/core/providers/anthropic/redactedthinking_test.go b/core/providers/anthropic/redactedthinking_test.go new file mode 100644 index 00000000000..2dae37b5ecd --- /dev/null +++ b/core/providers/anthropic/redactedthinking_test.go @@ -0,0 +1,299 @@ +package anthropic + +import ( + "context" + "encoding/json" + "testing" + + "github.com/maximhq/bifrost/core/schemas" +) + +// Anthropic returns redacted_thinking blocks when reasoning is flagged by its +// safety systems, and requires thinking and redacted_thinking blocks to be +// replayed unmodified on the next turn during tool use. These tests pin the +// chat-completions round trip of that payload: response conversion preserves +// it as a reasoning.encrypted detail, request conversion re-materializes the +// block, and streaming surfaces the payload from content_block_start. + +// A non-streaming response mixing a redacted_thinking block with a visible +// thinking block must yield two reasoning details in content order: a +// reasoning.encrypted entry carrying the data payload, then the signed text +// entry. The tool call must come through untouched. +func TestToBifrostChatResponse_PreservesRedactedThinking(t *testing.T) { + response := &AnthropicMessageResponse{ + ID: "msg_1", + Role: "assistant", + Model: "claude-sonnet-4-20250514", + Content: []AnthropicContentBlock{ + {Type: AnthropicContentBlockTypeRedactedThinking, Data: schemas.Ptr("ENCRYPTED_PAYLOAD")}, + {Type: AnthropicContentBlockTypeThinking, Thinking: schemas.Ptr("visible reasoning"), Signature: schemas.Ptr("sig-1")}, + {Type: AnthropicContentBlockTypeToolUse, ID: schemas.Ptr("toolu_1"), Name: schemas.Ptr("get_weather"), Input: json.RawMessage(`{"city":"paris"}`)}, + }, + StopReason: AnthropicStopReasonToolUse, + } + + ctx, cancel := schemas.NewBifrostContextWithCancel(context.Background()) + defer cancel() + result := response.ToBifrostChatResponse(ctx) + + if len(result.Choices) != 1 || result.Choices[0].Message == nil || result.Choices[0].Message.ChatAssistantMessage == nil { + t.Fatalf("unexpected choices: %+v", result.Choices) + } + details := result.Choices[0].Message.ChatAssistantMessage.ReasoningDetails + if len(details) != 2 { + t.Fatalf("expected 2 reasoning details (encrypted + text), got %d: %+v", len(details), details) + } + if details[0].Type != schemas.BifrostReasoningDetailsTypeEncrypted { + t.Errorf("details[0].Type = %q, want %q", details[0].Type, schemas.BifrostReasoningDetailsTypeEncrypted) + } + if details[0].Data == nil || *details[0].Data != "ENCRYPTED_PAYLOAD" { + t.Errorf("details[0].Data = %v, want ENCRYPTED_PAYLOAD", details[0].Data) + } + if details[0].Index != 0 || details[1].Index != 1 { + t.Errorf("detail indices = %d, %d, want 0, 1", details[0].Index, details[1].Index) + } + if details[1].Type != schemas.BifrostReasoningDetailsTypeText || details[1].Text == nil || *details[1].Text != "visible reasoning" { + t.Errorf("details[1] mangled: %+v", details[1]) + } + if details[1].Signature == nil || *details[1].Signature != "sig-1" { + t.Errorf("details[1].Signature = %v, want sig-1", details[1].Signature) + } +} + +// Redacted blocks with no data payload (absent or empty string) carry +// nothing to replay, so response conversion must not produce reasoning +// details for them. +func TestToBifrostChatResponse_RedactedThinkingWithoutDataSkipped(t *testing.T) { + response := &AnthropicMessageResponse{ + ID: "msg_1", + Role: "assistant", + Model: "claude-sonnet-4-20250514", + Content: []AnthropicContentBlock{ + {Type: AnthropicContentBlockTypeRedactedThinking}, + {Type: AnthropicContentBlockTypeRedactedThinking, Data: schemas.Ptr("")}, + {Type: AnthropicContentBlockTypeText, Text: schemas.Ptr("hello")}, + }, + StopReason: AnthropicStopReasonEndTurn, + } + + ctx, cancel := schemas.NewBifrostContextWithCancel(context.Background()) + defer cancel() + result := response.ToBifrostChatResponse(ctx) + + msg := result.Choices[0].Message + if msg.ChatAssistantMessage != nil && len(msg.ChatAssistantMessage.ReasoningDetails) != 0 { + t.Errorf("expected no reasoning details for data-less or empty-data redacted blocks, got %+v", msg.ChatAssistantMessage.ReasoningDetails) + } +} + +// Request conversion must re-materialize a reasoning.encrypted detail as a +// redacted_thinking block (data only, no thinking or signature fields), +// placed before the visible thinking block and the tool_use block, so the +// assistant turn is replayed exactly as the model produced it. +func TestToAnthropicChatRequest_ReplaysRedactedThinking(t *testing.T) { + toolID := "toolu_1" + bifrostReq := &schemas.BifrostChatRequest{ + Provider: schemas.Anthropic, + Model: "claude-sonnet-4-20250514", + Input: []schemas.ChatMessage{ + {Role: schemas.ChatMessageRoleUser, Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("weather in paris?")}}, + { + Role: schemas.ChatMessageRoleAssistant, + ChatAssistantMessage: &schemas.ChatAssistantMessage{ + ReasoningDetails: []schemas.ChatReasoningDetails{ + {Index: 0, Type: schemas.BifrostReasoningDetailsTypeEncrypted, Data: schemas.Ptr("ENCRYPTED_PAYLOAD")}, + {Index: 1, Type: schemas.BifrostReasoningDetailsTypeText, Text: schemas.Ptr("visible reasoning"), Signature: schemas.Ptr("sig-1")}, + }, + ToolCalls: []schemas.ChatAssistantMessageToolCall{ + {Index: 0, Type: schemas.Ptr("function"), ID: &toolID, + Function: schemas.ChatAssistantMessageToolCallFunction{Name: schemas.Ptr("get_weather"), Arguments: `{"city":"paris"}`}}, + }, + }, + }, + { + Role: schemas.ChatMessageRoleTool, + ChatToolMessage: &schemas.ChatToolMessage{ToolCallID: &toolID}, + Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("22C sunny")}, + }, + }, + Params: &schemas.ChatParameters{ + MaxCompletionTokens: schemas.Ptr(2048), + Reasoning: &schemas.ChatReasoning{MaxTokens: schemas.Ptr(1024)}, + }, + } + + ctx, cancel := schemas.NewBifrostContextWithCancel(context.Background()) + defer cancel() + result, err := ToAnthropicChatRequest(ctx, bifrostReq) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + + if len(result.Messages) < 2 { + t.Fatalf("expected at least 2 messages, got %d", len(result.Messages)) + } + assistant := result.Messages[1] + blocks := assistant.Content.ContentBlocks + if len(blocks) != 3 { + t.Fatalf("expected 3 assistant content blocks (redacted_thinking, thinking, tool_use), got %d: %+v", len(blocks), blocks) + } + if blocks[0].Type != AnthropicContentBlockTypeRedactedThinking { + t.Errorf("blocks[0].Type = %q, want redacted_thinking", blocks[0].Type) + } + if blocks[0].Data == nil || *blocks[0].Data != "ENCRYPTED_PAYLOAD" { + t.Errorf("blocks[0].Data = %v, want ENCRYPTED_PAYLOAD", blocks[0].Data) + } + if blocks[0].Thinking != nil || blocks[0].Signature != nil { + t.Errorf("redacted block must carry only data, got thinking=%v signature=%v", blocks[0].Thinking, blocks[0].Signature) + } + if blocks[1].Type != AnthropicContentBlockTypeThinking || blocks[1].Thinking == nil || *blocks[1].Thinking != "visible reasoning" { + t.Errorf("blocks[1] mangled: %+v", blocks[1]) + } + if blocks[2].Type != AnthropicContentBlockTypeToolUse { + t.Errorf("blocks[2].Type = %q, want tool_use", blocks[2].Type) + } +} + +func TestToAnthropicChatRequest_EncryptedDetailWithoutDataStaysThinking(t *testing.T) { + // Encrypted reasoning details produced by other providers (e.g. gemini + // thought signatures) carry a signature but no data, and a client may + // send an empty data string; both must keep the historical + // thinking-block mapping, not become redacted_thinking. + bifrostReq := &schemas.BifrostChatRequest{ + Provider: schemas.Anthropic, + Model: "claude-sonnet-4-20250514", + Input: []schemas.ChatMessage{ + {Role: schemas.ChatMessageRoleUser, Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("hi")}}, + { + Role: schemas.ChatMessageRoleAssistant, + Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("hello")}, + ChatAssistantMessage: &schemas.ChatAssistantMessage{ + ReasoningDetails: []schemas.ChatReasoningDetails{ + {Index: 0, Type: schemas.BifrostReasoningDetailsTypeEncrypted, Signature: schemas.Ptr("gemini-sig")}, + {Index: 1, Type: schemas.BifrostReasoningDetailsTypeEncrypted, Data: schemas.Ptr("")}, + }, + }, + }, + {Role: schemas.ChatMessageRoleUser, Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("and now?")}}, + }, + Params: &schemas.ChatParameters{MaxCompletionTokens: schemas.Ptr(2048)}, + } + + ctx, cancel := schemas.NewBifrostContextWithCancel(context.Background()) + defer cancel() + result, err := ToAnthropicChatRequest(ctx, bifrostReq) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + for _, block := range result.Messages[1].Content.ContentBlocks { + if block.Type == AnthropicContentBlockTypeRedactedThinking { + t.Errorf("encrypted detail without a payload must not become redacted_thinking: %+v", block) + } + } +} + +// A redacted_thinking content_block_start carries the complete payload (no +// deltas follow), so it must emit one chunk with a reasoning.encrypted +// detail. Starts with no payload, or for plain text blocks, stay silent. +func TestToBifrostChatCompletionStream_RedactedThinkingStart(t *testing.T) { + ctx, cancel := schemas.NewBifrostContextWithCancel(context.Background()) + defer cancel() + state := NewAnthropicStreamState() + + chunk := &AnthropicStreamEvent{ + Type: AnthropicStreamEventTypeContentBlockStart, + Index: schemas.Ptr(0), + ContentBlock: &AnthropicContentBlock{Type: AnthropicContentBlockTypeRedactedThinking, Data: schemas.Ptr("ENCRYPTED_PAYLOAD")}, + } + resp, bifrostErr, done := chunk.ToBifrostChatCompletionStream(ctx, "", state) + if bifrostErr != nil || done { + t.Fatalf("unexpected err=%v done=%v", bifrostErr, done) + } + if resp == nil { + t.Fatalf("redacted_thinking content_block_start produced no chunk") + } + delta := resp.Choices[0].ChatStreamResponseChoice.Delta + if len(delta.ReasoningDetails) != 1 { + t.Fatalf("expected 1 reasoning detail, got %+v", delta.ReasoningDetails) + } + if delta.ReasoningDetails[0].Type != schemas.BifrostReasoningDetailsTypeEncrypted { + t.Errorf("detail type = %q, want %q", delta.ReasoningDetails[0].Type, schemas.BifrostReasoningDetailsTypeEncrypted) + } + if delta.ReasoningDetails[0].Data == nil || *delta.ReasoningDetails[0].Data != "ENCRYPTED_PAYLOAD" { + t.Errorf("detail data = %v, want ENCRYPTED_PAYLOAD", delta.ReasoningDetails[0].Data) + } + + // A redacted block without data (or with empty data) and a plain text + // block start still produce no chunk. + for _, block := range []*AnthropicContentBlock{ + {Type: AnthropicContentBlockTypeRedactedThinking}, + {Type: AnthropicContentBlockTypeRedactedThinking, Data: schemas.Ptr("")}, + {Type: AnthropicContentBlockTypeText, Text: schemas.Ptr("")}, + } { + chunk := &AnthropicStreamEvent{Type: AnthropicStreamEventTypeContentBlockStart, Index: schemas.Ptr(1), ContentBlock: block} + resp, bifrostErr, done := chunk.ToBifrostChatCompletionStream(ctx, "", state) + if resp != nil || bifrostErr != nil || done { + t.Errorf("block %q: expected silent skip, got resp=%v err=%v done=%v", block.Type, resp, bifrostErr, done) + } + } +} + +// A stream mixing redacted and visible thinking blocks must keep one +// reasoning_details index per content block. The accumulator and replaying +// clients group reasoning deltas by that index; if a redacted block shares +// an index with a visible thinking block, the merged detail loses the +// encrypted type and the payload is dropped again on replay. +func TestToBifrostChatCompletionStream_MixedReasoningBlocksKeepDistinctIndices(t *testing.T) { + ctx, cancel := schemas.NewBifrostContextWithCancel(context.Background()) + defer cancel() + state := NewAnthropicStreamState() + + detailIndexes := func(resp *schemas.BifrostChatResponse) []int { + if resp == nil { + t.Fatalf("expected a chunk, got nil") + } + var idxs []int + for _, rd := range resp.Choices[0].ChatStreamResponseChoice.Delta.ReasoningDetails { + idxs = append(idxs, rd.Index) + } + return idxs + } + + // Block 0: redacted thinking. + resp, _, _ := (&AnthropicStreamEvent{ + Type: AnthropicStreamEventTypeContentBlockStart, + Index: schemas.Ptr(0), + ContentBlock: &AnthropicContentBlock{Type: AnthropicContentBlockTypeRedactedThinking, Data: schemas.Ptr("ENC_A")}, + }).ToBifrostChatCompletionStream(ctx, "", state) + if got := detailIndexes(resp); len(got) != 1 || got[0] != 0 { + t.Errorf("redacted block 0: detail indices = %v, want [0]", got) + } + + // Block 1: visible thinking, then its signature. + resp, _, _ = (&AnthropicStreamEvent{ + Type: AnthropicStreamEventTypeContentBlockDelta, + Index: schemas.Ptr(1), + Delta: &AnthropicStreamDelta{Type: AnthropicStreamDeltaTypeThinking, Thinking: schemas.Ptr("visible")}, + }).ToBifrostChatCompletionStream(ctx, "", state) + if got := detailIndexes(resp); len(got) != 1 || got[0] != 1 { + t.Errorf("thinking delta block 1: detail indices = %v, want [1]", got) + } + resp, _, _ = (&AnthropicStreamEvent{ + Type: AnthropicStreamEventTypeContentBlockDelta, + Index: schemas.Ptr(1), + Delta: &AnthropicStreamDelta{Type: AnthropicStreamDeltaTypeSignature, Signature: schemas.Ptr("sig-1")}, + }).ToBifrostChatCompletionStream(ctx, "", state) + if got := detailIndexes(resp); len(got) != 1 || got[0] != 1 { + t.Errorf("signature delta block 1: detail indices = %v, want [1]", got) + } + + // Block 2: a second redacted thinking block. + resp, _, _ = (&AnthropicStreamEvent{ + Type: AnthropicStreamEventTypeContentBlockStart, + Index: schemas.Ptr(2), + ContentBlock: &AnthropicContentBlock{Type: AnthropicContentBlockTypeRedactedThinking, Data: schemas.Ptr("ENC_B")}, + }).ToBifrostChatCompletionStream(ctx, "", state) + if got := detailIndexes(resp); len(got) != 1 || got[0] != 2 { + t.Errorf("redacted block 2: detail indices = %v, want [2]", got) + } +} diff --git a/core/providers/anthropic/requestbuilder.go b/core/providers/anthropic/requestbuilder.go index 0124bb7b072..20950d7cc2c 100644 --- a/core/providers/anthropic/requestbuilder.go +++ b/core/providers/anthropic/requestbuilder.go @@ -7,73 +7,123 @@ import ( "github.com/maximhq/bifrost/core/schemas" ) -// AnthropicRequestBuildConfig controls how BuildAnthropicResponsesRequestBody -// assembles the final JSON payload. Each Anthropic-family provider (Anthropic -// native, Azure, Vertex) fills in only the fields relevant to it and leaves -// the rest as zero values. +// AnthropicRequestBuildConfig holds the dynamic, per-call inputs to +// BuildAnthropic{Chat,Responses}RequestBody. The static, per-provider +// request-shaping flags (DeleteModelField, AddAnthropicVersion, etc.) live in +// AnthropicProviderRequestDefaultsMap and are looked up by Provider inside the +// builder — callers do not pass them. type AnthropicRequestBuildConfig struct { // Provider is used for feature-gating (field stripping, header injection, - // tool validation). Required. + // tool validation) and to look up static request-shaping defaults from + // AnthropicProviderRequestDefaultsMap. Required. Provider schemas.ModelProvider - // Deployment overrides the model field. When empty the model is read from - // the request and normalised via ParseModelString (Anthropic native path). - // Azure and Vertex set this to the deployment/model name. - Deployment string - - // DeleteModelField removes "model" from the output JSON body. - // Vertex passes model in the request URL, not the body. - // Ignored when IsCountTokens is true — count-tokens calls retain the model - // field so the provider can route to the correct endpoint. - DeleteModelField bool - - // DeleteRegionField removes the "region" field (Vertex only). - DeleteRegionField bool + // Model overrides the model field. When empty the model is read from + // the request. Azure, Vertex, and Bedrock set this to the deployment / + // model name. + Model string IsStreaming bool - // IsCountTokens enables token-counting mode (Vertex only): strips - // max_tokens and temperature from the body and keeps (or sets) the model - // field. + // IsCountTokens enables token-counting mode: strips max_tokens and + // temperature from the body and keeps (or sets) the model field. IsCountTokens bool - // ExcludeFields lists JSON top-level keys to remove from the final body in - // both the raw and typed paths. Used by Anthropic's count-tokens call to + // ExcludeFields lists JSON top-level keys to remove from the final body + // in both raw and typed paths. Used by Anthropic native count-tokens to // strip max_tokens and temperature after typed conversion. ExcludeFields []string + // ValidateTools runs ValidateToolsForProvider before typed conversion, + // returning an error for any tool unsupported by the provider. Set true + // on Responses-API paths (the chat API doesn't carry ResponsesTool types). + ValidateTools bool + + // BetaHeaderOverrides / ProviderExtraHeaders feed into the body-side + // anthropic_beta injection when the provider's defaults set + // InjectBetaHeadersIntoBody = true (Vertex only). Both come from the + // caller's NetworkConfig at request time. + BetaHeaderOverrides map[string]bool + ProviderExtraHeaders map[string]string + + // ShouldSendBackRawRequest / ShouldSendBackRawResponse control whether raw + // request/response bytes are attached to BifrostError.ExtraFields via + // providerUtils.EnrichError. + ShouldSendBackRawRequest bool + ShouldSendBackRawResponse bool +} + +// AnthropicProviderRequestDefaults captures the static, per-provider request- +// shaping flags applied by BuildAnthropic{Chat,Responses}RequestBody. +// +// Keep this in lockstep with ProviderFeatures (utils.go) — together they +// describe everything an Anthropic-family provider needs for request shaping. +type AnthropicProviderRequestDefaults struct { + // DeleteModelField removes "model" from the output JSON body. Used by + // providers that put model in the URL (Vertex, Bedrock). Ignored when + // IsCountTokens is true — those calls retain model for routing. + DeleteModelField bool + + // DeleteRegionField removes the "region" field from the body (Vertex only). + DeleteRegionField bool + + // DeleteStreamField removes "stream" from the body unconditionally. + // Bedrock determines streaming via URL endpoint (invoke vs + // invoke-with-response-stream); the body must never carry a stream field. + DeleteStreamField bool + // AddAnthropicVersion injects "anthropic_version" into the body when the - // field is absent (Vertex only). + // field is absent (Vertex, Bedrock). AddAnthropicVersion bool AnthropicVersion string // StripCacheControlScope calls SetStripCacheControlScope(true) on the - // typed request struct before marshalling (Vertex typed path). + // typed request struct before marshalling (Vertex only). StripCacheControlScope bool - // RemapToolVersions runs RemapRawToolVersionsForProvider on the raw body to - // downgrade unsupported tool type versions (Vertex raw path). + // RemapToolVersions runs RemapRawToolVersionsForProvider on the body to + // downgrade unsupported tool type versions (Vertex, Bedrock). RemapToolVersions bool // InjectBetaHeadersIntoBody serialises filtered beta headers into the JSON - // body as "anthropic_beta". Vertex embeds beta headers in the body rather - // than HTTP request headers. + // body as "anthropic_beta" (Vertex only — embeds in body, others use HTTP). InjectBetaHeadersIntoBody bool - BetaHeaderOverrides map[string]bool - ProviderExtraHeaders map[string]string - - // ValidateTools runs ValidateResponsesToolsForProvider before typed - // conversion, silently dropping any tool unsupported by the provider (Azure, - // Vertex) so the request still reaches the provider without it. Mirrors the - // Chat path's strip-silently policy. - ValidateTools bool +} - // ShouldSendBackRawRequest / ShouldSendBackRawResponse control whether raw - // request/response bytes are attached to BifrostError.ExtraFields via - // providerUtils.EnrichError. Vertex honours per-provider send-back flags; - // Anthropic and Azure leave both false. - ShouldSendBackRawRequest bool - ShouldSendBackRawResponse bool +// AnthropicProviderRequestDefaultsMap maps each Anthropic-family provider to +// the static request-shaping defaults it needs. The builder reads from this +// map directly using cfg.Provider — callers do not set these fields. +var AnthropicProviderRequestDefaultsMap = map[schemas.ModelProvider]AnthropicProviderRequestDefaults{ + schemas.Anthropic: {}, + schemas.Azure: {}, + // Bedrock Mantle native-Anthropic endpoint (/anthropic/v1/messages): the + // request is the native Anthropic Messages body, so model stays in the body + // (set to the bare Bedrock model id), the version is sent as an + // "anthropic-version" HTTP header rather than a body field, and stream is a + // body field. Tool type versions are still remapped to the canonical pair + // the hosted Claude generation expects. + schemas.Bedrock: { + RemapToolVersions: true, + }, + // Bedrock Mantle shares the Bedrock native-Anthropic request shape (model in + // body, anthropic-version HTTP header, tool versions remapped). It has its own + // entry so its feature surface in ProviderFeatures can diverge from Bedrock's + // Converse path without coupling the two. + schemas.BedrockMantle: { + RemapToolVersions: true, + }, + // Vertex publisher endpoint: model + region in URL, anthropic_version + // required, beta headers in body (not HTTP), cache_control.scope stripped + // at marshal time, tool versions remapped. + schemas.Vertex: { + DeleteModelField: true, + DeleteRegionField: true, + AddAnthropicVersion: true, + AnthropicVersion: "vertex-2023-10-16", + StripCacheControlScope: true, + RemapToolVersions: true, + InjectBetaHeadersIntoBody: true, + }, } // BuildAnthropicResponsesRequestBody is the single implementation of the @@ -90,6 +140,8 @@ func BuildAnthropicResponsesRequestBody(ctx *schemas.BifrostContext, request *sc // capModel is the canonical model used for capability gating in the raw-body capModel := schemas.ResolveCanonicalModel(ctx, request.Model) + defaults := AnthropicProviderRequestDefaultsMap[cfg.Provider] + newErr := func(msg string, err error, reqBody []byte) *schemas.BifrostError { return providerUtils.EnrichError( ctx, @@ -117,22 +169,22 @@ func BuildAnthropicResponsesRequestBody(ctx *schemas.BifrostContext, request *sc if err != nil { return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) } - jsonBody, err = providerUtils.SetJSONField(jsonBody, "model", cfg.Deployment) + jsonBody, err = providerUtils.SetJSONField(jsonBody, "model", cfg.Model) if err != nil { return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) } } else { // Normal path: handle model field per provider. - if cfg.Deployment != "" { - if cfg.DeleteModelField { - // Vertex: model lives in the URL. + if cfg.Model != "" { + if defaults.DeleteModelField { + // Vertex/Bedrock: model lives in the URL. jsonBody, err = providerUtils.DeleteJSONField(jsonBody, "model") if err != nil { return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) } } else { // Azure: replace model with deployment name. - jsonBody, err = providerUtils.SetJSONField(jsonBody, "model", cfg.Deployment) + jsonBody, err = providerUtils.SetJSONField(jsonBody, "model", cfg.Model) if err != nil { return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) } @@ -147,7 +199,7 @@ func BuildAnthropicResponsesRequestBody(ctx *schemas.BifrostContext, request *sc // Ensure max_tokens is present. if !providerUtils.JSONFieldExists(jsonBody, "max_tokens") { - modelForTokens := cfg.Deployment + modelForTokens := cfg.Model if modelForTokens == "" { if r := providerUtils.GetJSONField(jsonBody, "model"); r.Exists() { modelForTokens = r.String() @@ -175,7 +227,7 @@ func BuildAnthropicResponsesRequestBody(ctx *schemas.BifrostContext, request *sc return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) } - if cfg.RemapToolVersions { + if defaults.RemapToolVersions { // request.Model is the alias-resolved model id; pass it so // computer-use / text-editor / bash tools get normalized to the // canonical {type, name} pair Anthropic expects for the model's generation. @@ -185,7 +237,7 @@ func BuildAnthropicResponsesRequestBody(ctx *schemas.BifrostContext, request *sc } } - if cfg.DeleteRegionField { + if defaults.DeleteRegionField { jsonBody, err = providerUtils.DeleteJSONField(jsonBody, "region") if err != nil { return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) @@ -197,8 +249,8 @@ func BuildAnthropicResponsesRequestBody(ctx *schemas.BifrostContext, request *sc return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) } - if cfg.AddAnthropicVersion && !providerUtils.JSONFieldExists(jsonBody, "anthropic_version") { - jsonBody, err = providerUtils.SetJSONField(jsonBody, "anthropic_version", cfg.AnthropicVersion) + if defaults.AddAnthropicVersion && !providerUtils.JSONFieldExists(jsonBody, "anthropic_version") { + jsonBody, err = providerUtils.SetJSONField(jsonBody, "anthropic_version", defaults.AnthropicVersion) if err != nil { return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) } @@ -242,11 +294,11 @@ func BuildAnthropicResponsesRequestBody(ctx *schemas.BifrostContext, request *sc return nil, newErr("request body is not provided", nil, jsonBody) } - if cfg.Deployment != "" { - reqBody.Model = cfg.Deployment + if cfg.Model != "" { + reqBody.Model = cfg.Model } - if cfg.StripCacheControlScope { + if defaults.StripCacheControlScope { reqBody.SetStripCacheControlScope(true) } @@ -254,6 +306,13 @@ func BuildAnthropicResponsesRequestBody(ctx *schemas.BifrostContext, request *sc reqBody.Stream = schemas.Ptr(true) } + // Strip request- and tool-level fields the target provider doesn't + // support. ToAnthropicResponsesRequest doesn't do this internally + // (unlike ToAnthropicChatRequest), so the builder must — keeping + // behaviour symmetric across raw and typed paths and across both + // chat/responses APIs. + stripUnsupportedAnthropicFields(reqBody, cfg.Provider, request.Model) + AddMissingBetaHeadersToContext(ctx, reqBody, cfg.Provider) jsonBody, err = providerUtils.MarshalSorted(reqBody) @@ -271,8 +330,8 @@ func BuildAnthropicResponsesRequestBody(ctx *schemas.BifrostContext, request *sc } } - if cfg.AddAnthropicVersion && !providerUtils.JSONFieldExists(jsonBody, "anthropic_version") { - jsonBody, err = providerUtils.SetJSONField(jsonBody, "anthropic_version", cfg.AnthropicVersion) + if defaults.AddAnthropicVersion && !providerUtils.JSONFieldExists(jsonBody, "anthropic_version") { + jsonBody, err = providerUtils.SetJSONField(jsonBody, "anthropic_version", defaults.AnthropicVersion) if err != nil { return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) } @@ -287,15 +346,15 @@ func BuildAnthropicResponsesRequestBody(ctx *schemas.BifrostContext, request *sc if err != nil { return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) } - } else if cfg.DeleteModelField { - // Vertex: model is in the URL, remove it from the body. + } else if defaults.DeleteModelField { + // Vertex/Bedrock: model is in the URL, remove it from the body. jsonBody, err = providerUtils.DeleteJSONField(jsonBody, "model") if err != nil { return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) } } - if cfg.DeleteRegionField { + if defaults.DeleteRegionField { jsonBody, err = providerUtils.DeleteJSONField(jsonBody, "region") if err != nil { return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) @@ -320,7 +379,235 @@ func BuildAnthropicResponsesRequestBody(ctx *schemas.BifrostContext, request *sc return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) } - if cfg.InjectBetaHeadersIntoBody { + if defaults.DeleteStreamField { + jsonBody, err = providerUtils.DeleteJSONField(jsonBody, "stream") + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + } + + if defaults.InjectBetaHeadersIntoBody { + if betaHeaders := FilterBetaHeadersForProvider(MergeBetaHeaders(ctx, cfg.ProviderExtraHeaders), cfg.Provider, cfg.BetaHeaderOverrides); len(betaHeaders) > 0 { + jsonBody, err = providerUtils.SetJSONField(jsonBody, "anthropic_beta", betaHeaders) + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + } + } + + return jsonBody, nil +} + +// BuildAnthropicChatRequestBody is the chat-completion analogue of +// BuildAnthropicResponsesRequestBody, shared by Anthropic, Azure, Vertex, and +// Bedrock for ChatCompletion / ChatCompletionStream paths. It mirrors the +// responses pipeline (raw vs typed branching, field stripping, beta-header +// injection, fallbacks deletion) but operates on BifrostChatRequest + +// ToAnthropicChatRequest. IsCountTokens is not honoured here — count-tokens +// is a Responses-API concept. +func BuildAnthropicChatRequestBody(ctx *schemas.BifrostContext, request *schemas.BifrostChatRequest, cfg AnthropicRequestBuildConfig) ([]byte, *schemas.BifrostError) { + if providerUtils.IsLargePayloadPassthroughEnabled(ctx) { + return nil, nil + } + + // capModel is the canonical model used for capability gating in the raw-body + // path; the wire request.Model may be an opaque alias/deployment id. + capModel := schemas.ResolveCanonicalModel(ctx, request.Model) + + defaults := AnthropicProviderRequestDefaultsMap[cfg.Provider] + + newErr := func(msg string, err error, reqBody []byte) *schemas.BifrostError { + return providerUtils.EnrichError( + ctx, + providerUtils.NewBifrostOperationError(msg, err), + reqBody, + nil, + cfg.ShouldSendBackRawRequest, + cfg.ShouldSendBackRawResponse, + ) + } + + var jsonBody []byte + var err error + + if useRawBody, ok := ctx.Value(schemas.BifrostContextKeyUseRawRequestBody).(bool); ok && useRawBody { + jsonBody = request.GetRawRequestBody() + + if cfg.Model != "" { + if defaults.DeleteModelField { + jsonBody, err = providerUtils.DeleteJSONField(jsonBody, "model") + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + } else { + jsonBody, err = providerUtils.SetJSONField(jsonBody, "model", cfg.Model) + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + } + } else { + jsonBody, err = providerUtils.SetJSONField(jsonBody, "model", request.Model) + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + } + + if !providerUtils.JSONFieldExists(jsonBody, "max_tokens") { + modelForTokens := cfg.Model + if modelForTokens == "" { + if r := providerUtils.GetJSONField(jsonBody, "model"); r.Exists() { + modelForTokens = r.String() + } + } + jsonBody, err = providerUtils.SetJSONField(jsonBody, "max_tokens", providerUtils.GetMaxOutputTokensOrDefault(modelForTokens, AnthropicDefaultMaxTokens)) + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + } + + if cfg.IsStreaming { + jsonBody, err = providerUtils.SetJSONField(jsonBody, "stream", true) + } else { + jsonBody, err = providerUtils.DeleteJSONField(jsonBody, "stream") + } + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + + jsonBody, err = StripAutoInjectableTools(jsonBody) + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + + if defaults.RemapToolVersions { + jsonBody, err = RemapRawToolVersionsForProvider(jsonBody, cfg.Provider, capModel) + if err != nil { + return nil, newErr(err.Error(), nil, jsonBody) + } + } + + if defaults.DeleteRegionField { + jsonBody, err = providerUtils.DeleteJSONField(jsonBody, "region") + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + } + + jsonBody, err = StripUnsupportedFieldsFromRawBody(jsonBody, cfg.Provider, capModel) + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + + if defaults.AddAnthropicVersion && !providerUtils.JSONFieldExists(jsonBody, "anthropic_version") { + jsonBody, err = providerUtils.SetJSONField(jsonBody, "anthropic_version", defaults.AnthropicVersion) + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + } + + var probe AnthropicMessageRequest + if unmarshalErr := schemas.Unmarshal(jsonBody, &probe); unmarshalErr == nil { + AddMissingBetaHeadersToContext(ctx, &probe, cfg.Provider) + } + + for _, field := range cfg.ExcludeFields { + jsonBody, err = providerUtils.DeleteJSONField(jsonBody, field) + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + } + } else { + reqBody, convErr := ToAnthropicChatRequest(ctx, request) + if convErr != nil { + return nil, newErr(schemas.ErrRequestBodyConversion, convErr, jsonBody) + } + if reqBody == nil { + return nil, newErr("request body is not provided", nil, jsonBody) + } + + if cfg.Model != "" { + reqBody.Model = cfg.Model + } + + if defaults.StripCacheControlScope { + reqBody.SetStripCacheControlScope(true) + } + + if cfg.IsStreaming { + reqBody.Stream = schemas.Ptr(true) + } + + // Re-strip with cfg.Provider (canonical) in case the request was + // routed through a custom-provider alias whose name doesn't match + // the ProviderFeatures map entry. Idempotent — ToAnthropicChatRequest + // already strips using bifrostReq.Provider, so this only changes + // behaviour when the two diverge. + stripUnsupportedAnthropicFields(reqBody, cfg.Provider, request.Model) + + AddMissingBetaHeadersToContext(ctx, reqBody, cfg.Provider) + + jsonBody, err = providerUtils.MarshalSorted(reqBody) + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, fmt.Errorf("failed to marshal request body: %w", err), jsonBody) + } + + if ctx.Value(schemas.BifrostContextKeyPassthroughExtraParams) == true { + extraParams := reqBody.GetExtraParams() + if len(extraParams) > 0 { + jsonBody, err = providerUtils.MergeExtraParamsIntoJSON(jsonBody, extraParams) + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + } + } + + if defaults.AddAnthropicVersion && !providerUtils.JSONFieldExists(jsonBody, "anthropic_version") { + jsonBody, err = providerUtils.SetJSONField(jsonBody, "anthropic_version", defaults.AnthropicVersion) + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + } + + if defaults.DeleteModelField { + jsonBody, err = providerUtils.DeleteJSONField(jsonBody, "model") + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + } + + if defaults.DeleteRegionField { + jsonBody, err = providerUtils.DeleteJSONField(jsonBody, "region") + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + } + + for _, field := range cfg.ExcludeFields { + jsonBody, err = providerUtils.DeleteJSONField(jsonBody, field) + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + } + } + + jsonBody, err = StripEmptyThinkingBlocks(jsonBody) + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + + jsonBody, err = providerUtils.DeleteJSONField(jsonBody, "fallbacks") + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + + if defaults.DeleteStreamField { + jsonBody, err = providerUtils.DeleteJSONField(jsonBody, "stream") + if err != nil { + return nil, newErr(schemas.ErrProviderRequestMarshal, err, jsonBody) + } + } + + if defaults.InjectBetaHeadersIntoBody { if betaHeaders := FilterBetaHeadersForProvider(MergeBetaHeaders(ctx, cfg.ProviderExtraHeaders), cfg.Provider, cfg.BetaHeaderOverrides); len(betaHeaders) > 0 { jsonBody, err = providerUtils.SetJSONField(jsonBody, "anthropic_beta", betaHeaders) if err != nil { diff --git a/core/providers/anthropic/requestbuilder_test.go b/core/providers/anthropic/requestbuilder_test.go index a7dea8548f2..71b209096fc 100644 --- a/core/providers/anthropic/requestbuilder_test.go +++ b/core/providers/anthropic/requestbuilder_test.go @@ -88,9 +88,8 @@ func TestBuildAnthropicResponsesRequestBody_RawBodyPath(t *testing.T) { } result, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ - Provider: schemas.Vertex, - Deployment: "claude-sonnet-4-5", - DeleteModelField: true, + Provider: schemas.Vertex, + Model: "claude-sonnet-4-5", }) if err != nil { t.Fatalf("unexpected error: %v", err) @@ -112,8 +111,8 @@ func TestBuildAnthropicResponsesRequestBody_RawBodyPath(t *testing.T) { } result, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ - Provider: schemas.Azure, - Deployment: "my-azure-deployment", + Provider: schemas.Azure, + Model: "my-azure-deployment", }) if err != nil { t.Fatalf("unexpected error: %v", err) @@ -141,8 +140,8 @@ func TestBuildAnthropicResponsesRequestBody_RawBodyPath(t *testing.T) { } result, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ - Provider: schemas.Azure, - Deployment: "my-azure-deployment", + Provider: schemas.Azure, + Model: "my-azure-deployment", }) if err != nil { t.Fatalf("unexpected error: %v", err) @@ -213,10 +212,8 @@ func TestBuildAnthropicResponsesRequestBody_RawBodyPath(t *testing.T) { } result, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ - Provider: schemas.Vertex, - Deployment: "claude-sonnet-4-5", - DeleteModelField: true, - DeleteRegionField: true, + Provider: schemas.Vertex, + Model: "claude-sonnet-4-5", }) if err != nil { t.Fatalf("unexpected error: %v", err) @@ -238,11 +235,8 @@ func TestBuildAnthropicResponsesRequestBody_RawBodyPath(t *testing.T) { } result, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ - Provider: schemas.Vertex, - Deployment: "claude-sonnet-4-5", - DeleteModelField: true, - AddAnthropicVersion: true, - AnthropicVersion: "vertex-2023-10-16", + Provider: schemas.Vertex, + Model: "claude-sonnet-4-5", }) if err != nil { t.Fatalf("unexpected error: %v", err) @@ -313,10 +307,8 @@ func TestBuildAnthropicResponsesRequestBody_RawBodyPath(t *testing.T) { } result, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ - Provider: schemas.Vertex, - Deployment: "claude-sonnet-4-5", - DeleteModelField: true, - InjectBetaHeadersIntoBody: true, + Provider: schemas.Vertex, + Model: "claude-sonnet-4-5", }) if err != nil { t.Fatalf("unexpected error: %v", err) @@ -341,7 +333,7 @@ func TestBuildAnthropicResponsesRequestBody_CountTokensMode(t *testing.T) { result, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ Provider: schemas.Vertex, - Deployment: "claude-sonnet-4-5", + Model: "claude-sonnet-4-5", IsCountTokens: true, }) if err != nil { @@ -371,7 +363,7 @@ func TestBuildAnthropicResponsesRequestBody_CountTokensMode(t *testing.T) { result, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ Provider: schemas.Vertex, - Deployment: "new-deployment", + Model: "new-deployment", IsCountTokens: true, }) if err != nil { @@ -443,9 +435,8 @@ func TestBuildAnthropicResponsesRequestBody_TypedPath(t *testing.T) { } result, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ - Provider: schemas.Vertex, - Deployment: "claude-sonnet-4-5", - DeleteModelField: true, + Provider: schemas.Vertex, + Model: "claude-sonnet-4-5", }) if err != nil { t.Fatalf("unexpected error: %v", err) @@ -466,11 +457,8 @@ func TestBuildAnthropicResponsesRequestBody_TypedPath(t *testing.T) { } result, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ - Provider: schemas.Vertex, - Deployment: "claude-sonnet-4-5", - DeleteModelField: true, - AddAnthropicVersion: true, - AnthropicVersion: "vertex-2023-10-16", + Provider: schemas.Vertex, + Model: "claude-sonnet-4-5", }) if err != nil { t.Fatalf("unexpected error: %v", err) @@ -497,7 +485,7 @@ func TestBuildAnthropicResponsesRequestBody_TypedPath(t *testing.T) { result, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ Provider: schemas.Vertex, - Deployment: "claude-sonnet-4-5", + Model: "claude-sonnet-4-5", IsCountTokens: true, }) if err != nil { @@ -589,6 +577,7 @@ func TestDoesWebSearchOrFetchAutoInjectCodeExecution(t *testing.T) { {string(AnthropicToolTypeWebFetch20250910), false}, {string(AnthropicToolTypeWebFetch20260209), true}, {string(AnthropicToolTypeWebFetch20260309), true}, + {string(AnthropicToolTypeWebFetch20260318), true}, {"web_search_unknown", true}, {"web_fetch_unknown", true}, {"unknown_type", true}, @@ -741,9 +730,8 @@ func TestBuildAnthropicResponsesRequestBody_StripCacheControlScope(t *testing.T) } _, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ - Provider: schemas.Vertex, - Deployment: "claude-sonnet-4-5", - StripCacheControlScope: true, + Provider: schemas.Vertex, + Model: "claude-sonnet-4-5", }) if err != nil { t.Fatalf("unexpected error: %v", err) @@ -763,10 +751,8 @@ func TestBuildAnthropicResponsesRequestBody_RemapToolVersions(t *testing.T) { } result, err := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ - Provider: schemas.Vertex, - Deployment: "claude-sonnet-4-5", - DeleteModelField: true, - RemapToolVersions: true, + Provider: schemas.Vertex, + Model: "claude-sonnet-4-5", }) if err != nil { t.Fatalf("unexpected error: %v", err) diff --git a/core/providers/anthropic/responses.go b/core/providers/anthropic/responses.go index e79fbc3bf42..89ed9bf7413 100644 --- a/core/providers/anthropic/responses.go +++ b/core/providers/anthropic/responses.go @@ -33,8 +33,10 @@ type AnthropicResponsesStreamState struct { WebSearchCaller *AnthropicToolCaller // Programmatic-tool-calling caller, if the search was spawned from code execution // Web fetch tool accumulation - WebFetchToolID *string // Tool ID of active web fetch - WebFetchOutputIndex *int // Output index for this fetch + WebFetchToolID *string // Tool ID of active web fetch + WebFetchOutputIndex *int // Output index for this fetch + WebFetchURL *string // URL captured from the server_tool_use input + WebFetchResult *AnthropicContentBlock // Result block when it arrives // Advisor tool accumulation AdvisorToolID *string // Tool ID of active advisor call @@ -165,6 +167,26 @@ type anthropicToResponsesStreamState struct { nextBlockIndex int blockIndexByItem map[string]int + // blockIndexMisses records non-empty item keys for which blockIndexFor was + // asked for an index that allocBlockIndex never assigned — i.e. a + // content_block_stop/_delta referencing a block whose content_block_start was + // never registered at output_item.added. This must not happen: every + // output_item.added allocates its block index (~L2053), and the Anthropic + // passthrough path routes every output_item.added through this converter so the + // allocator always runs (see mustConvertInPassthrough in the transport). A miss + // therefore signals a stream-bookkeeping desync; it is recorded so tests fail + // loudly instead of the stream silently mis-numbering blocks. + blockIndexMisses []string + + // passthrough is true when this reverse conversion runs on the Claude Code + // passthrough path, where verbatim raw upstream frames are interleaved with the + // converter's own. Only then must the converter consume a content-block index for + // server-tool result blocks it still collapses (currently resultless web_search) + // to stay in lockstep with the upstream indices carried by interleaved raw frames. + // On the all-normalized path (OpenAI-via-Anthropic, curl), every frame is + // converter-built, so consuming the index would skip a number. + passthrough bool + // codeExecToolNameByItem remembers each code_interpreter_call's Anthropic // sub-tool name (captured at output_item.added) so the code block's input can be // reconstructed and closed on code.done — emitted before the nested web_search @@ -196,14 +218,20 @@ func (s *anthropicToResponsesStreamState) allocBlockIndex(key string) *int { return &idx } -// blockIndexFor returns the index previously allocated for key, allocating a fresh -// one if the block's start was never seen (defensive — keeps start/stop paired). +// blockIndexFor returns the index previously allocated for key. A non-empty key +// that was never registered by allocBlockIndex means a stop/delta references a +// block whose start we never emitted — a bookkeeping desync — so it is recorded +// in blockIndexMisses before falling back to a fresh allocation (defensive — keeps +// start/stop paired rather than crashing the stream). func (s *anthropicToResponsesStreamState) blockIndexFor(key string) *int { if key != "" && s.blockIndexByItem != nil { if idx, ok := s.blockIndexByItem[key]; ok { return &idx } } + if key != "" { + s.blockIndexMisses = append(s.blockIndexMisses, key) + } return s.allocBlockIndex(key) } @@ -240,8 +268,16 @@ func getOrCreateAnthropicToResponsesStreamState(ctx *schemas.BifrostContext) *an return state } -// acquireAnthropicResponsesStreamState gets an Anthropic responses stream state from the pool. -func acquireAnthropicResponsesStreamState() *AnthropicResponsesStreamState { +// SetResponsesStreamPassthrough marks this request's Anthropic reverse stream +// conversion as running on the Claude Code passthrough path (raw upstream frames +// interleaved with converted ones). The converter reads state.passthrough only for +// collapsed server-tool result blocks that must still consume an upstream index. +func SetResponsesStreamPassthrough(ctx *schemas.BifrostContext) { + getOrCreateAnthropicToResponsesStreamState(ctx).passthrough = true +} + +// AcquireAnthropicResponsesStreamState gets an Anthropic responses stream state from the pool. +func AcquireAnthropicResponsesStreamState() *AnthropicResponsesStreamState { state := anthropicResponsesStreamStatePool.Get().(*AnthropicResponsesStreamState) // Clear maps (they're already initialized from New or previous flush) // Only initialize if nil (shouldn't happen, but defensive) @@ -319,6 +355,8 @@ func acquireAnthropicResponsesStreamState() *AnthropicResponsesStreamState { state.WebSearchCaller = nil state.WebFetchToolID = nil state.WebFetchOutputIndex = nil + state.WebFetchURL = nil + state.WebFetchResult = nil state.AdvisorToolID = nil state.AdvisorOutputIndex = nil state.AdvisorResult = nil @@ -343,8 +381,8 @@ func acquireAnthropicResponsesStreamState() *AnthropicResponsesStreamState { return state } -// releaseAnthropicResponsesStreamState returns an Anthropic responses stream state to the pool. -func releaseAnthropicResponsesStreamState(state *AnthropicResponsesStreamState) { +// ReleaseAnthropicResponsesStreamState returns an Anthropic responses stream state to the pool. +func ReleaseAnthropicResponsesStreamState(state *AnthropicResponsesStreamState) { if state != nil { state.flush() // Clean before returning to pool anthropicResponsesStreamStatePool.Put(state) @@ -363,6 +401,8 @@ func (state *AnthropicResponsesStreamState) flush() { state.WebSearchCaller = nil state.WebFetchToolID = nil state.WebFetchOutputIndex = nil + state.WebFetchURL = nil + state.WebFetchResult = nil state.AdvisorToolID = nil state.AdvisorOutputIndex = nil state.AdvisorResult = nil @@ -658,16 +698,37 @@ func (chunk *AnthropicStreamEvent) ToBifrostResponsesStream(ctx context.Context, state.beginInputJSONBuffer(chunk.Index, anthropicInputJSONBufferWebFetch) state.WebFetchToolID = chunk.ContentBlock.ID state.WebFetchOutputIndex = schemas.Ptr(outputIndex) + state.WebFetchURL = nil + if u := providerUtils.GetJSONField(chunk.ContentBlock.Input, "url"); u.Exists() && u.Type == gjson.String { + state.WebFetchURL = schemas.Ptr(u.Str) + } state.ItemIDs[outputIndex] = *chunk.ContentBlock.ID + toolMsg := &schemas.ResponsesToolMessage{ + CallID: chunk.ContentBlock.ID, + } + if state.WebFetchURL != nil { + toolMsg.Action = &schemas.ResponsesToolMessageActionStruct{ + ResponsesWebFetchToolCallAction: &schemas.ResponsesWebFetchToolCallAction{ + Type: "fetch", + URL: *state.WebFetchURL, + }, + } + } + // Preserve the programmatic-tool-calling caller (set when this fetch was + // spawned from inside the code-execution sandbox), mirroring web_search. + if chunk.ContentBlock.Caller != nil { + toolMsg.Caller = &schemas.ResponsesToolCaller{ + Type: string(chunk.ContentBlock.Caller.Type), + ToolID: chunk.ContentBlock.Caller.ToolID, + } + } item := &schemas.ResponsesMessage{ - ID: chunk.ContentBlock.ID, - Type: schemas.Ptr(schemas.ResponsesMessageTypeWebFetchCall), - Status: schemas.Ptr("in_progress"), - ResponsesToolMessage: &schemas.ResponsesToolMessage{ - CallID: chunk.ContentBlock.ID, - }, + ID: chunk.ContentBlock.ID, + Type: schemas.Ptr(schemas.ResponsesMessageTypeWebFetchCall), + Status: schemas.Ptr("in_progress"), + ResponsesToolMessage: toolMsg, } var responses []*schemas.BifrostResponsesStreamResponse @@ -710,6 +771,12 @@ func (chunk *AnthropicStreamEvent) ToBifrostResponsesStream(ctx context.Context, delete(state.ContentIndexToBlockType, *chunk.Index) } + // Remember that the result block arrived; its content is not + // represented in the Responses model (handled server-side), but its + // presence drives the web_fetch_call output_item.done at the result + // block's content_block_stop (mirrors web_search). + state.WebFetchResult = chunk.ContentBlock + return []*schemas.BifrostResponsesStreamResponse{{ Type: schemas.ResponsesStreamResponseTypeWebFetchCallCompleted, SequenceNumber: sequenceNumber, @@ -1326,6 +1393,14 @@ func (chunk *AnthropicStreamEvent) ToBifrostResponsesStream(ctx context.Context, return nil, nil, false case anthropicInputJSONBufferWebFetch: + // Fallback: recover the URL if Anthropic streamed it via + // input_json_delta instead of inlining it in content_block_start + // (mirrors the web_search query fallback above). + if state.WebFetchURL == nil && inputJSON != "" { + if u := providerUtils.GetJSONField([]byte(inputJSON), "url"); u.Exists() && u.Type == gjson.String { + state.WebFetchURL = schemas.Ptr(u.Str) + } + } return nil, nil, false case anthropicInputJSONBufferAdvisor: @@ -1467,6 +1542,48 @@ func (chunk *AnthropicStreamEvent) ToBifrostResponsesStream(ctx context.Context, }, nil, false } + // End of a web_fetch_tool_result block — emit the web_fetch_call done with + // the typed result payload so Anthropic-compatible reverse conversion can + // faithfully rebuild the server_tool_use + web_fetch_tool_result pair. + if state.WebFetchResult != nil && state.WebFetchToolID != nil { + toolMsg := &schemas.ResponsesToolMessage{CallID: state.WebFetchToolID} + if state.WebFetchURL != nil { + toolMsg.Action = &schemas.ResponsesToolMessageActionStruct{ + ResponsesWebFetchToolCallAction: &schemas.ResponsesWebFetchToolCallAction{ + Type: "fetch", + URL: *state.WebFetchURL, + }, + } + } + toolMsg.ResponsesWebFetchCall = convertAnthropicWebFetchResultToBifrost(state.WebFetchResult) + item := &schemas.ResponsesMessage{ + ID: state.WebFetchToolID, + Type: schemas.Ptr(schemas.ResponsesMessageTypeWebFetchCall), + Status: schemas.Ptr("completed"), + ResponsesToolMessage: toolMsg, + } + outputIdx := state.WebFetchOutputIndex + + state.WebFetchToolID = nil + state.WebFetchOutputIndex = nil + state.WebFetchURL = nil + state.WebFetchResult = nil + + if chunk.Index != nil { + delete(state.ContentIndexToBlockType, *chunk.Index) + } + + return []*schemas.BifrostResponsesStreamResponse{ + { + Type: schemas.ResponsesStreamResponseTypeOutputItemDone, + SequenceNumber: sequenceNumber, + OutputIndex: outputIdx, + ContentIndex: chunk.Index, + Item: item, + }, + }, nil, false + } + // End of an advisor_tool_result block — emit the advisor_call done. if state.AdvisorResult != nil && state.AdvisorToolID != nil { advisor := &schemas.ResponsesAdvisorCall{} @@ -2156,6 +2273,18 @@ func ToAnthropicResponsesStreamResponse(ctx *schemas.BifrostContext, bifrostResp Name: schemas.Ptr(toolName), Input: json.RawMessage("{}"), } + } else if bifrostResp.Item != nil && + bifrostResp.Item.Type != nil && + *bifrostResp.Item.Type == schemas.ResponsesMessageTypeWebFetchCall { + + // Web fetch call - emit content_block_start with server_tool_use. The + // paired web_fetch_tool_result block is emitted at output_item.done when + // the typed result payload is available. + streamResp.Type = AnthropicStreamEventTypeContentBlockStart + streamResp.Index = blockIdx + if blocks := convertBifrostWebFetchCallToAnthropicBlocks(bifrostResp.Item); len(blocks) > 0 { + streamResp.ContentBlock = &blocks[0] + } } else { // Text or other content blocks - emit content_block_start streamResp.Type = AnthropicStreamEventTypeContentBlockStart @@ -2457,6 +2586,13 @@ func ToAnthropicResponsesStreamResponse(ctx *schemas.BifrostContext, bifrostResp } } } + return []*AnthropicStreamEvent{ + streamResp, + { + Type: AnthropicStreamEventTypeContentBlockStop, + Index: streamResp.Index, + }, + } } else if bifrostResp.Item != nil && bifrostResp.Item.Type != nil && *bifrostResp.Item.Type == schemas.ResponsesMessageTypeWebSearchCall { @@ -2515,8 +2651,41 @@ func ToAnthropicResponsesStreamResponse(ctx *schemas.BifrostContext, bifrostResp &AnthropicStreamEvent{Type: AnthropicStreamEventTypeContentBlockStart, Index: resultIndex, ContentBlock: resultBlock}, &AnthropicStreamEvent{Type: AnthropicStreamEventTypeContentBlockStop, Index: resultIndex}, ) + } else if state.passthrough { + // A resultless or error web_search (no sources, max_uses_exceeded, + // web_search_tool_result_error, etc.) still consumed a content-block + // index upstream. On the passthrough path, later non-server-tool frames + // may be forwarded raw, so consume that hidden index to keep the + // converter's next allocated index in lockstep with upstream. + _ = state.allocBlockIndex("") } + return events + } else if bifrostResp.Item != nil && + bifrostResp.Item.Type != nil && + *bifrostResp.Item.Type == schemas.ResponsesMessageTypeWebFetchCall { + + // Web fetch call complete - close the server_tool_use block, then emit + // the paired web_fetch_tool_result block when present. + state := getOrCreateAnthropicToResponsesStreamState(ctx) + serverIdx := state.blockIndexFor(reverseStreamItemKey(bifrostResp)) + events := []*AnthropicStreamEvent{{ + Type: AnthropicStreamEventTypeContentBlockStop, + Index: serverIdx, + }} + if blocks := convertBifrostWebFetchCallToAnthropicBlocks(bifrostResp.Item); len(blocks) > 1 { + resultIdx := state.allocBlockIndex("") + resultBlock := blocks[1] + events = append(events, + &AnthropicStreamEvent{Type: AnthropicStreamEventTypeContentBlockStart, Index: resultIdx, ContentBlock: &resultBlock}, + &AnthropicStreamEvent{Type: AnthropicStreamEventTypeContentBlockStop, Index: resultIdx}, + ) + } else if state.passthrough { + // Older/partial replay items can lack the typed web_fetch result even + // though upstream consumed a result-block index. Keep later raw + // passthrough frames aligned with the converter's allocator. + _ = state.allocBlockIndex("") + } return events } else if bifrostResp.Item != nil && bifrostResp.Item.Type != nil && @@ -3038,8 +3207,9 @@ func ToAnthropicResponsesRequest(ctx *schemas.BifrostContext, bifrostReq *schema } } if bifrostReq.Params.Text != nil { - // Vertex doesn't support native structured outputs, so convert to tool - if bifrostReq.Provider == schemas.Vertex { + // Vertex and Bedrock Mantle don't accept native structured outputs + // (output_config.format), so convert to a tool instead. + if bifrostReq.Provider == schemas.Vertex || bifrostReq.Provider == schemas.BedrockMantle { if bifrostReq.Params.Text.Format != nil { responseFormatTool := convertResponsesTextFormatToTool(ctx, bifrostReq.Params.Text) if responseFormatTool != nil { @@ -4227,49 +4397,26 @@ func ConvertBifrostMessagesToAnthropicMessages(ctx *schemas.BifrostContext, bifr case schemas.ResponsesMessageTypeWebFetchCall: flushPendingToolResults() - - if currentAssistantMessage == nil { - currentAssistantMessage = &AnthropicMessage{ - Role: AnthropicMessageRoleAssistant, - } - } - - if len(pendingReasoningContentBlocks) > 0 { - copied := make([]AnthropicContentBlock, len(pendingReasoningContentBlocks)) - copy(copied, pendingReasoningContentBlocks) - pendingToolCalls = append(copied, pendingToolCalls...) - pendingReasoningContentBlocks = nil - } - - serverToolUseBlock := AnthropicContentBlock{ - Type: AnthropicContentBlockTypeServerToolUse, - Name: schemas.Ptr(string(AnthropicToolNameWebFetch)), - } - if msg.ResponsesToolMessage != nil && msg.ResponsesToolMessage.Caller != nil { - serverToolUseBlock.Caller = &AnthropicToolCaller{ - Type: AnthropicToolCallerType(msg.ResponsesToolMessage.Caller.Type), - ToolID: msg.ResponsesToolMessage.Caller.ToolID, + webFetchBlocks := convertBifrostWebFetchCallToAnthropicBlocks(&msg) + if len(webFetchBlocks) > 0 { + if currentAssistantMessage == nil { + currentAssistantMessage = &AnthropicMessage{ + Role: AnthropicMessageRoleAssistant, + } } - } - if msg.ID != nil { - serverToolUseBlock.ID = msg.ID - } - if msg.ResponsesToolMessage != nil && msg.ResponsesToolMessage.Action != nil && - msg.ResponsesToolMessage.Action.ResponsesWebFetchToolCallAction != nil { - inputBytes, err := providerUtils.MarshalSorted(map[string]interface{}{ - "url": msg.ResponsesToolMessage.Action.ResponsesWebFetchToolCallAction.URL, - }) - if err == nil { - serverToolUseBlock.Input = json.RawMessage(inputBytes) + if len(pendingReasoningContentBlocks) > 0 { + copied := make([]AnthropicContentBlock, len(pendingReasoningContentBlocks)) + copy(copied, pendingReasoningContentBlocks) + pendingToolCalls = append(copied, pendingToolCalls...) + pendingReasoningContentBlocks = nil } - } - pendingToolCalls = append(pendingToolCalls, serverToolUseBlock) - - if serverToolUseBlock.ID != nil { - if currentToolCallIDs == nil { - currentToolCallIDs = make(map[string]bool) + pendingToolCalls = append(pendingToolCalls, webFetchBlocks...) + if webFetchBlocks[0].ID != nil { + if currentToolCallIDs == nil { + currentToolCallIDs = make(map[string]bool) + } + currentToolCallIDs[*webFetchBlocks[0].ID] = true } - currentToolCallIDs[*serverToolUseBlock.ID] = true } // Handle other tool call types that are not natively supported by Anthropic @@ -4586,7 +4733,6 @@ func convertAnthropicContentBlocksToResponsesMessagesGrouped(contentBlocks []Ant } bifrostMsg.ResponsesToolMessage.Output.ResponsesFunctionToolCallOutputBlocks = toolMsgContentBlocks } - // Handle is_error from Anthropic if block.IsError != nil && *block.IsError { bifrostMsg.Status = schemas.Ptr("incomplete") @@ -4947,7 +5093,6 @@ func convertAnthropicContentBlocksToResponsesMessages(ctx *schemas.BifrostContex } bifrostMsg.ResponsesToolMessage.Output.ResponsesFunctionToolCallOutputBlocks = toolMsgContentBlocks } - // Handle is_error from Anthropic if block.IsError != nil && *block.IsError { bifrostMsg.Status = schemas.Ptr("incomplete") @@ -4997,7 +5142,7 @@ func convertAnthropicContentBlocksToResponsesMessages(ctx *schemas.BifrostContex bifrostMsg := schemas.ResponsesMessage{ Type: schemas.Ptr(schemas.ResponsesMessageTypeWebFetchCall), Status: schemas.Ptr("completed"), - ResponsesToolMessage: &schemas.ResponsesToolMessage{}, + ResponsesToolMessage: &schemas.ResponsesToolMessage{CallID: block.ID}, } if block.Caller != nil { @@ -5092,7 +5237,9 @@ func convertAnthropicContentBlocksToResponsesMessages(ctx *schemas.BifrostContex } case AnthropicContentBlockTypeWebFetchToolResult: - // Web fetch results are handled server-side by Anthropic, skip + if block.ToolUseID != nil { + attachAnthropicWebFetchResult(bifrostMessages, *block.ToolUseID, block) + } case AnthropicContentBlockTypeWebSearchToolResultError: // Handle web search errors — find matching web_search_call and mark as failed @@ -5682,6 +5829,162 @@ func convertBifrostWebSearchCallToAnthropicBlocks(msg *schemas.ResponsesMessage) return blocks } +func getAnthropicContentObject(content *AnthropicContent) *AnthropicContentBlock { + if content == nil { + return nil + } + if content.ContentObj != nil { + return content.ContentObj + } + if len(content.ContentBlocks) > 0 { + return &content.ContentBlocks[0] + } + return nil +} + +func convertAnthropicWebFetchResultToBifrost(block *AnthropicContentBlock) *schemas.ResponsesWebFetchCall { + if block == nil { + return nil + } + result := &schemas.ResponsesWebFetchCall{} + inner := getAnthropicContentObject(block.Content) + if inner == nil { + return result + } + + result.ResultType = string(inner.Type) + result.URL = inner.URL + result.RetrievedAt = inner.RetrievedAt + result.ErrorCode = inner.ErrorCode + + doc := getAnthropicContentObject(inner.Content) + if doc != nil { + result.Document = &schemas.ResponsesWebFetchDocument{ + Type: string(doc.Type), + Text: doc.Text, + Title: doc.Title, + Context: doc.Context, + } + if doc.Citations != nil && doc.Citations.Config != nil { + result.Document.Citations = doc.Citations.Config + } + if doc.Source != nil && doc.Source.SourceObj != nil { + src := doc.Source.SourceObj + result.Document.Source = &schemas.ResponsesWebFetchSource{ + Type: src.Type, + MediaType: src.MediaType, + Data: src.Data, + URL: src.URL, + FileID: src.FileID, + } + } + } + + return result +} + +func attachAnthropicWebFetchResult(messages []schemas.ResponsesMessage, toolUseID string, block AnthropicContentBlock) { + for i := len(messages) - 1; i >= 0; i-- { + msg := &messages[i] + if msg.Type == nil || *msg.Type != schemas.ResponsesMessageTypeWebFetchCall { + continue + } + if msg.ID == nil || *msg.ID != toolUseID { + continue + } + if msg.ResponsesToolMessage == nil { + msg.ResponsesToolMessage = &schemas.ResponsesToolMessage{} + } + msg.ResponsesToolMessage.CallID = &toolUseID + msg.ResponsesToolMessage.ResponsesWebFetchCall = convertAnthropicWebFetchResultToBifrost(&block) + break + } +} + +func convertBifrostWebFetchCallToAnthropicBlocks(msg *schemas.ResponsesMessage) []AnthropicContentBlock { + if msg == nil || msg.ResponsesToolMessage == nil { + return nil + } + tm := msg.ResponsesToolMessage + + var toolUseID *string + if tm.CallID != nil { + toolUseID = tm.CallID + } else { + toolUseID = msg.ID + } + + var caller *AnthropicToolCaller + if tm.Caller != nil { + caller = &AnthropicToolCaller{Type: AnthropicToolCallerType(tm.Caller.Type), ToolID: tm.Caller.ToolID} + } + + serverToolUseBlock := AnthropicContentBlock{ + Type: AnthropicContentBlockTypeServerToolUse, + ID: toolUseID, + Name: schemas.Ptr(string(AnthropicToolNameWebFetch)), + Caller: caller, + Input: json.RawMessage("{}"), + } + if tm.Action != nil && tm.Action.ResponsesWebFetchToolCallAction != nil && + tm.Action.ResponsesWebFetchToolCallAction.URL != "" { + if inputBytes, err := providerUtils.MarshalSorted(map[string]interface{}{"url": tm.Action.ResponsesWebFetchToolCallAction.URL}); err == nil { + serverToolUseBlock.Input = json.RawMessage(inputBytes) + } + } + + blocks := []AnthropicContentBlock{serverToolUseBlock} + if tm.ResponsesWebFetchCall == nil { + return blocks + } + + wf := tm.ResponsesWebFetchCall + resultType := wf.ResultType + if resultType == "" { + resultType = "web_fetch_result" + } + inner := AnthropicContentBlock{ + Type: AnthropicContentBlockType(resultType), + URL: wf.URL, + RetrievedAt: wf.RetrievedAt, + ErrorCode: wf.ErrorCode, + } + if wf.Document != nil { + doc := &AnthropicContentBlock{ + Type: AnthropicContentBlockType(wf.Document.Type), + Text: wf.Document.Text, + Title: wf.Document.Title, + Context: wf.Document.Context, + } + if doc.Type == "" { + doc.Type = AnthropicContentBlockTypeDocument + } + if wf.Document.Citations != nil { + doc.Citations = &AnthropicCitations{Config: wf.Document.Citations} + } + if wf.Document.Source != nil { + src := wf.Document.Source + doc.Source = &AnthropicBlockSource{SourceObj: &AnthropicSource{ + Type: src.Type, + MediaType: src.MediaType, + Data: src.Data, + URL: src.URL, + FileID: src.FileID, + }} + } + inner.Content = &AnthropicContent{ContentObj: doc} + } + + blocks = append(blocks, AnthropicContentBlock{ + Type: AnthropicContentBlockTypeWebFetchToolResult, + ToolUseID: toolUseID, + Caller: caller, + Content: &AnthropicContent{ContentObj: &inner}, + }) + + return blocks +} + // convertBifrostAdvisorCallToAnthropicBlocks rebuilds the advisor server_tool_use // block and its paired advisor_tool_result block from a neutral advisor_call. // Anthropic requires both blocks to appear together in the assistant message. @@ -6133,28 +6436,14 @@ func convertAnthropicToolToBifrost(tool *AnthropicTool) *schemas.ResponsesTool { // Handle special tool types first if tool.Type != nil { - switch *tool.Type { - case AnthropicToolTypeComputer20250124, AnthropicToolTypeComputer20251124: - bifrostTool := &schemas.ResponsesTool{ - Type: schemas.ResponsesToolTypeComputerUsePreview, - } - if tool.AnthropicToolComputerUse != nil { - bifrostTool.ResponsesToolComputerUsePreview = &schemas.ResponsesToolComputerUsePreview{ - Environment: "browser", // Default environment - } - if tool.AnthropicToolComputerUse.DisplayWidthPx != nil { - bifrostTool.ResponsesToolComputerUsePreview.DisplayWidth = *tool.AnthropicToolComputerUse.DisplayWidthPx - } - if tool.AnthropicToolComputerUse.DisplayHeightPx != nil { - bifrostTool.ResponsesToolComputerUsePreview.DisplayHeight = *tool.AnthropicToolComputerUse.DisplayHeightPx - } - if tool.AnthropicToolComputerUse.EnableZoom != nil { - bifrostTool.ResponsesToolComputerUsePreview.EnableZoom = tool.AnthropicToolComputerUse.EnableZoom - } - } - return bifrostTool - - case AnthropicToolTypeWebSearch20250305, AnthropicToolTypeWebSearch20260209: + // Version-dated server search tools ship new versions regularly; match them by + // prefix so a newer version (e.g. web_fetch_20260318) is recognized as a server + // tool instead of falling through to the client-function default at the end of + // this function. Mirrors applySharedServerToolFields and the chat path. Tools + // that must round-trip their exact version (code_execution, text_editor) stay in + // the exact switch below. + switch typeStr := string(*tool.Type); { + case strings.HasPrefix(typeStr, "web_search_"): bifrostTool := &schemas.ResponsesTool{ Type: schemas.ResponsesToolTypeWebSearch, } @@ -6180,14 +6469,16 @@ func convertAnthropicToolToBifrost(tool *AnthropicTool) *schemas.ResponsesTool { return bifrostTool - case AnthropicToolTypeWebFetch20250910, AnthropicToolTypeWebFetch20260209, AnthropicToolTypeWebFetch20260309: + case strings.HasPrefix(typeStr, "web_fetch_"): bifrostTool := &schemas.ResponsesTool{ Type: schemas.ResponsesToolTypeWebFetch, } if tool.AnthropicToolWebFetch != nil { bifrostTool.ResponsesToolWebFetch = &schemas.ResponsesToolWebFetch{ - MaxUses: tool.AnthropicToolWebFetch.MaxUses, - MaxContentTokens: tool.AnthropicToolWebFetch.MaxContentTokens, + MaxUses: tool.AnthropicToolWebFetch.MaxUses, + MaxContentTokens: tool.AnthropicToolWebFetch.MaxContentTokens, + UseCache: tool.AnthropicToolWebFetch.UseCache, + ResponseInclusion: tool.AnthropicToolWebFetch.ResponseInclusion, } if len(tool.AnthropicToolWebFetch.AllowedDomains) > 0 || len(tool.AnthropicToolWebFetch.BlockedDomains) > 0 { bifrostTool.ResponsesToolWebFetch.Filters = &schemas.ResponsesToolWebSearchFilters{ @@ -6197,6 +6488,28 @@ func convertAnthropicToolToBifrost(tool *AnthropicTool) *schemas.ResponsesTool { } } return bifrostTool + } + + switch *tool.Type { + case AnthropicToolTypeComputer20250124, AnthropicToolTypeComputer20251124: + bifrostTool := &schemas.ResponsesTool{ + Type: schemas.ResponsesToolTypeComputerUsePreview, + } + if tool.AnthropicToolComputerUse != nil { + bifrostTool.ResponsesToolComputerUsePreview = &schemas.ResponsesToolComputerUsePreview{ + Environment: "browser", // Default environment + } + if tool.AnthropicToolComputerUse.DisplayWidthPx != nil { + bifrostTool.ResponsesToolComputerUsePreview.DisplayWidth = *tool.AnthropicToolComputerUse.DisplayWidthPx + } + if tool.AnthropicToolComputerUse.DisplayHeightPx != nil { + bifrostTool.ResponsesToolComputerUsePreview.DisplayHeight = *tool.AnthropicToolComputerUse.DisplayHeightPx + } + if tool.AnthropicToolComputerUse.EnableZoom != nil { + bifrostTool.ResponsesToolComputerUsePreview.EnableZoom = tool.AnthropicToolComputerUse.EnableZoom + } + } + return bifrostTool case AnthropicToolTypeCodeExecution20250522, AnthropicToolTypeCodeExecution, AnthropicToolTypeCodeExecution20260120, AnthropicToolTypeCodeExecution20260521: @@ -6579,6 +6892,12 @@ func convertBifrostToolToAnthropic(model string, tool *schemas.ResponsesTool, pr (strings.Contains(model, "4.6") || strings.Contains(model, "4-6")) { webFetchType = AnthropicToolTypeWebFetch20260309 } + if tool.ResponsesToolWebFetch != nil && tool.ResponsesToolWebFetch.ResponseInclusion != nil { + webFetchType = AnthropicToolTypeWebFetch20260318 + } else if tool.ResponsesToolWebFetch != nil && tool.ResponsesToolWebFetch.UseCache != nil && + webFetchType == AnthropicToolTypeWebFetch20250910 { + webFetchType = AnthropicToolTypeWebFetch20260309 + } anthropicTool := &AnthropicTool{ Type: schemas.Ptr(webFetchType), Name: string(AnthropicToolNameWebFetch), @@ -6587,6 +6906,8 @@ func convertBifrostToolToAnthropic(model string, tool *schemas.ResponsesTool, pr if tool.ResponsesToolWebFetch != nil { anthropicTool.AnthropicToolWebFetch.MaxUses = tool.ResponsesToolWebFetch.MaxUses anthropicTool.AnthropicToolWebFetch.MaxContentTokens = tool.ResponsesToolWebFetch.MaxContentTokens + anthropicTool.AnthropicToolWebFetch.UseCache = tool.ResponsesToolWebFetch.UseCache + anthropicTool.AnthropicToolWebFetch.ResponseInclusion = tool.ResponsesToolWebFetch.ResponseInclusion if tool.ResponsesToolWebFetch.Filters != nil { anthropicTool.AnthropicToolWebFetch.AllowedDomains = tool.ResponsesToolWebFetch.Filters.AllowedDomains anthropicTool.AnthropicToolWebFetch.BlockedDomains = tool.ResponsesToolWebFetch.Filters.BlockedDomains @@ -6793,10 +7114,11 @@ func convertContentBlockToAnthropic(block schemas.ResponsesMessageContentBlock) } } case schemas.ResponsesInputMessageContentBlockTypeFile: - if block.ResponsesInputMessageContentBlockFile != nil { + if block.ResponsesInputMessageContentBlockFile != nil || block.FileID != nil { // Direct conversion without intermediate ChatContentBlock anthropicBlock := ConvertResponsesFileBlockToAnthropic( block.ResponsesInputMessageContentBlockFile, + block.FileID, block.CacheControl, block.Citations, ) @@ -6892,6 +7214,11 @@ func (block AnthropicContentBlock) toBifrostResponsesDocumentBlock() schemas.Res resultBlock.ResponsesInputMessageContentBlockFile.FileType = schemas.Ptr("text/plain") resultBlock.ResponsesInputMessageContentBlockFile.FileData = src.Data } + case "file": + // File ID reference (requires files-api-2025-04-14 beta header) + if src.FileID != nil { + resultBlock.FileID = src.FileID + } } return resultBlock @@ -7489,4 +7816,4 @@ func generateSyntheticInputJSONDeltas(argumentsJSON string, contentIndex *int) [ } return events -} +} \ No newline at end of file diff --git a/core/providers/anthropic/types.go b/core/providers/anthropic/types.go index 3dec8976a13..91b8ea897be 100644 --- a/core/providers/anthropic/types.go +++ b/core/providers/anthropic/types.go @@ -200,14 +200,8 @@ var ProviderFeatures = map[schemas.ModelProvider]ProviderFeatureSupport{ WebSearchNova: true, // nova_grounding — Responses path only CodeExecNova: true, // nova_code_interpreter — Responses path only ComputerUse: true, Bash: true, Memory: true, TextEditor: true, ToolSearch: true, - ContainerBasic: true, - // StructuredOutputs: kept true to match pre-existing behavior and the - // provider_feature_support_test.go assertion, but NEITHER B-header - // NOR B-platform upstream docs document strict tool validation / - // output_format on Bedrock. Needs live verification. If Bedrock's - // Converse API actually rejects `strict: true`, flip this to false - // and update the corresponding test assertion. - StructuredOutputs: true, + ContainerBasic: true, + StructuredOutputs: true, // documented on Bedrock per A overview matrix Compaction: true, // compact-2026-01-12 per B-header ContextEditing: true, // context-management-2025-06-27 per B-header (bundles memory) ContextManagementField: true, // Bedrock accepts context_management body field @@ -220,6 +214,32 @@ var ProviderFeatures = map[schemas.ModelProvider]ProviderFeatureSupport{ // narrow tool-examples-2025-10-29 header is, gated via InputExamples above. ServiceTier: true, // Bedrock handles service_tier via its own typed conversion }, + // Bedrock Mantle — same AWS-hosted Claude models as Bedrock, reached through + // the native Anthropic Messages surface (/anthropic/v1/messages) instead of + // Converse. Feature support is a property of the model+cloud, so this mirrors + // schemas.Bedrock — with one deliberate exception: the *Nova flags below. + // + // WebSearchNova / CodeExecNova are intentionally OFF here. They exist only to + // keep web_search / code_interpreter tools so the Bedrock Converse/Responses + // converter can rewrite them into nova_grounding / nova_code_interpreter. + // Mantle uses the native Anthropic body builder, which never runs that + // conversion, so leaving them on would forward an un-rewritten web_search / + // code_interpreter tool that the endpoint rejects. Mantle's native surface + // does not support the Anthropic web_search / code_execution server tools + // either, so both stay false (no WebSearch / CodeExecution). + schemas.BedrockMantle: { + ComputerUse: true, Bash: true, Memory: true, TextEditor: true, ToolSearch: true, + ContainerBasic: true, + StructuredOutputs: true, + Compaction: true, + ContextEditing: true, + ContextManagementField: true, + InterleavedThinking: true, + Context1M: true, + EagerInputStreaming: true, + InputExamples: true, + ServiceTier: true, + }, // Microsoft Azure AI Foundry — cite: A (most features azureAiBeta) + // Az-platform ("supports most of Claude's features"). Excluded per // Az-platform: Admin API, Models API, Message Batch API (not in scope). @@ -374,8 +394,8 @@ type AnthropicMessageRequest struct { ServiceTier *string `json:"service_tier,omitempty"` // "auto" or "standard_only" InferenceGeo *string `json:"inference_geo,omitempty"` // the geographic region for inference processing. If not specified, the workspace's default_inference_geo is used. ContextManagement *ContextManagement `json:"context_management,omitempty"` - Container *AnthropicContainer `json:"container,omitempty"` // string id OR object with skills[]; skills require skills-2025-10-02 beta - Diagnostics *AnthropicDiagnostics `json:"diagnostics,omitempty"` // cache diagnostics opt-in; requires cache-diagnosis-2026-04-07 beta (Anthropic API only) + Container *AnthropicContainer `json:"container,omitempty"` // string id OR object with skills[]; skills require skills-2025-10-02 beta + Diagnostics *AnthropicDiagnostics `json:"diagnostics,omitempty"` // cache diagnostics opt-in; requires cache-diagnosis-2026-04-07 beta (Anthropic API only) // Extra params for advanced use cases ExtraParams map[string]interface{} `json:"-"` @@ -919,12 +939,12 @@ const ( // code_execution inner result-content discriminators (the "content" object on // a *_code_execution_tool_result block; ContentObj.Type carries these). - AnthropicContentBlockTypeCodeExecutionResult AnthropicContentBlockType = "code_execution_result" // legacy Python (code_execution) - AnthropicContentBlockTypeEncryptedCodeExecutionResult AnthropicContentBlockType = "encrypted_code_execution_result" // code_execution with encrypted stdout - AnthropicContentBlockTypeBashCodeExecutionResult AnthropicContentBlockType = "bash_code_execution_result" // bash_code_execution - AnthropicContentBlockTypeTextEditorCodeExecutionResult AnthropicContentBlockType = "text_editor_code_execution_result" // text_editor_code_execution - AnthropicContentBlockTypeCodeExecutionToolResultError AnthropicContentBlockType = "code_execution_tool_result_error" // legacy Python error - AnthropicContentBlockTypeBashCodeExecutionToolResultError AnthropicContentBlockType = "bash_code_execution_tool_result_error" // bash error + AnthropicContentBlockTypeCodeExecutionResult AnthropicContentBlockType = "code_execution_result" // legacy Python (code_execution) + AnthropicContentBlockTypeEncryptedCodeExecutionResult AnthropicContentBlockType = "encrypted_code_execution_result" // code_execution with encrypted stdout + AnthropicContentBlockTypeBashCodeExecutionResult AnthropicContentBlockType = "bash_code_execution_result" // bash_code_execution + AnthropicContentBlockTypeTextEditorCodeExecutionResult AnthropicContentBlockType = "text_editor_code_execution_result" // text_editor_code_execution + AnthropicContentBlockTypeCodeExecutionToolResultError AnthropicContentBlockType = "code_execution_tool_result_error" // legacy Python error + AnthropicContentBlockTypeBashCodeExecutionToolResultError AnthropicContentBlockType = "bash_code_execution_tool_result_error" // bash error AnthropicContentBlockTypeTextEditorCodeExecutionResultError AnthropicContentBlockType = "text_editor_code_execution_tool_result_error" // code_execution file-output blocks (inside a result's "content" array; carry file_id). AnthropicContentBlockTypeCodeExecutionOutput AnthropicContentBlockType = "code_execution_output" // legacy Python output file @@ -1236,6 +1256,7 @@ const ( AnthropicToolTypeWebFetch20250910 AnthropicToolType = "web_fetch_20250910" AnthropicToolTypeWebFetch20260209 AnthropicToolType = "web_fetch_20260209" // Dynamic filtering AnthropicToolTypeWebFetch20260309 AnthropicToolType = "web_fetch_20260309" + AnthropicToolTypeWebFetch20260318 AnthropicToolType = "web_fetch_20260318" // Memory (client-side) AnthropicToolTypeMemory20250818 AnthropicToolType = "memory_20250818" @@ -1269,9 +1290,9 @@ const ( AnthropicToolNameBashCodeExecution AnthropicToolName = "bash_code_execution" AnthropicToolNameTextEditorCodeExecution AnthropicToolName = "text_editor_code_execution" AnthropicToolNameMemory AnthropicToolName = "memory" - AnthropicToolNameToolSearchBM25 AnthropicToolName = "tool_search_tool_bm25" - AnthropicToolNameToolSearchRegex AnthropicToolName = "tool_search_tool_regex" - AnthropicToolNameAdvisor AnthropicToolName = "advisor" + AnthropicToolNameToolSearchBM25 AnthropicToolName = "tool_search_tool_bm25" + AnthropicToolNameToolSearchRegex AnthropicToolName = "tool_search_tool_regex" + AnthropicToolNameAdvisor AnthropicToolName = "advisor" ) type AnthropicToolComputerUse struct { @@ -1297,12 +1318,13 @@ type AnthropicToolWebSearch struct { } type AnthropicToolWebFetch struct { - MaxUses *int `json:"max_uses,omitempty"` - AllowedDomains []string `json:"allowed_domains,omitempty"` - BlockedDomains []string `json:"blocked_domains,omitempty"` - MaxContentTokens *int `json:"max_content_tokens,omitempty"` - Citations *AnthropicCitations `json:"citations,omitempty"` // {enabled: bool} — toggles citation emission on fetched documents - UseCache *bool `json:"use_cache,omitempty"` // web_fetch_20260309+ only — enables server-side page cache + MaxUses *int `json:"max_uses,omitempty"` + AllowedDomains []string `json:"allowed_domains,omitempty"` + BlockedDomains []string `json:"blocked_domains,omitempty"` + MaxContentTokens *int `json:"max_content_tokens,omitempty"` + Citations *AnthropicCitations `json:"citations,omitempty"` // {enabled: bool} — toggles citation emission on fetched documents + UseCache *bool `json:"use_cache,omitempty"` // web_fetch_20260309+ only — enables server-side page cache + ResponseInclusion *string `json:"response_inclusion,omitempty"` // web_fetch_20260318+ only — "full" | "excluded" } // AnthropicToolTextEditor holds fields specific to the text_editor tool @@ -1717,9 +1739,10 @@ type AnthropicStreamError struct { // AnthropicFileUploadRequest represents a request to upload a file. type AnthropicFileUploadRequest struct { - File []byte `json:"-"` // Raw file content (not serialized) - Filename string `json:"filename"` // Original filename - Purpose string `json:"purpose"` // Purpose of the file (e.g., "batch") + File []byte `json:"-"` // Raw file content (not serialized) + Filename string `json:"filename"` // Original filename + Purpose string `json:"purpose"` // Purpose of the file (e.g., "batch") + ContentType *string `json:"content_type,omitempty"` // MIME type of the file } // AnthropicFileRetrieveRequest represents a request to retrieve a file. @@ -1838,4 +1861,4 @@ func parseAnthropicFileTimestamp(timestamp string) int64 { // AnthropicCountTokensResponse models the payload returned by Anthropic's count tokens endpoint. type AnthropicCountTokensResponse struct { InputTokens int `json:"input_tokens"` -} +} \ No newline at end of file diff --git a/core/providers/anthropic/utils.go b/core/providers/anthropic/utils.go index 932644f7a11..2b9a2fc9658 100644 --- a/core/providers/anthropic/utils.go +++ b/core/providers/anthropic/utils.go @@ -1029,18 +1029,6 @@ func setEffortOnOutputConfig(req *AnthropicMessageRequest, effort string) { req.OutputConfig.Effort = &effort } -// getRequestBodyForResponses serializes a BifrostResponsesRequest into the Anthropic wire format. -// It delegates to BuildAnthropicResponsesRequestBody with the appropriate provider and streaming config. -func getRequestBodyForResponses(ctx *schemas.BifrostContext, request *schemas.BifrostResponsesRequest, isStreaming bool, excludeFields []string, shouldSendBackRawRequest bool, shouldSendBackRawResponse bool) ([]byte, *schemas.BifrostError) { - return BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ - Provider: schemas.Anthropic, - IsStreaming: isStreaming, - ExcludeFields: excludeFields, - ShouldSendBackRawRequest: shouldSendBackRawRequest, - ShouldSendBackRawResponse: shouldSendBackRawResponse, - }) -} - // AddMissingBetaHeadersToContext analyzes the Anthropic request and adds missing beta headers to the context. // The provider parameter controls which headers are included — unsupported headers for the given provider are skipped. func AddMissingBetaHeadersToContext(ctx *schemas.BifrostContext, req *AnthropicMessageRequest, provider schemas.ModelProvider) error { @@ -1216,6 +1204,26 @@ func AddMissingBetaHeadersToContext(ctx *schemas.BifrostContext, req *AnthropicM } } } + // Check for file_id references (document/image blocks with a "file" + // source), which require the Files API beta header. + hasFileSource := false + for _, message := range req.Messages { + if hasFileSource { + break + } + if message.Content.ContentBlocks == nil { + continue + } + for _, block := range message.Content.ContentBlocks { + if block.Source != nil && block.Source.SourceObj != nil && block.Source.SourceObj.Type == "file" { + if !hasProvider || features.FilesAPI { + headers = appendUniqueHeader(headers, AnthropicFilesAPIBetaHeader) + } + hasFileSource = true + break + } + } + } if len(headers) == 0 { return nil } @@ -1353,6 +1361,8 @@ func doesWebSearchOrFetchAutoInjectCodeExecution(toolType string) bool { return true case string(AnthropicToolTypeWebFetch20260309): return true + case string(AnthropicToolTypeWebFetch20260318): + return true case string(AnthropicToolTypeWebFetch20250910): return false case string(AnthropicToolTypeWebFetch20260209): @@ -1368,6 +1378,10 @@ func doesWebSearchOrFetchAutoInjectCodeExecution(toolType string) bool { // with an empty "signature" field. An empty signature means the block came // from a non-Anthropic upstream (OpenAI never emits signatures; Anthropic // always does), so it is unsafe to replay to Anthropic. +// +// The predicate must stay scoped to "thinking" blocks: "redacted_thinking" +// blocks carry only an encrypted "data" payload (no thinking or signature +// fields) and must be replayed to Anthropic untouched. func StripEmptyThinkingBlocks(jsonBody []byte) ([]byte, error) { messagesResult := providerUtils.GetJSONField(jsonBody, "messages") if !messagesResult.Exists() || !messagesResult.IsArray() { @@ -2196,6 +2210,13 @@ func ConvertToAnthropicDocumentBlock(block schemas.ChatContentBlock) AnthropicCo documentBlock.Title = file.Filename } + // Handle uploaded file references from OpenAI-compatible file blocks. + if file.FileID != nil && *file.FileID != "" { + documentBlock.Source.SourceObj.Type = "file" + documentBlock.Source.SourceObj.FileID = file.FileID + return documentBlock + } + // Handle file URL if file.FileURL != nil && *file.FileURL != "" { documentBlock.Source.SourceObj.Type = "url" @@ -2252,7 +2273,7 @@ func ConvertToAnthropicDocumentBlock(block schemas.ChatContentBlock) AnthropicCo } // ConvertResponsesFileBlockToAnthropic converts a Responses file block directly to Anthropic document format -func ConvertResponsesFileBlockToAnthropic(fileBlock *schemas.ResponsesInputMessageContentBlockFile, cacheControl *schemas.CacheControl, citations *schemas.Citations) AnthropicContentBlock { +func ConvertResponsesFileBlockToAnthropic(fileBlock *schemas.ResponsesInputMessageContentBlockFile, fileID *string, cacheControl *schemas.CacheControl, citations *schemas.Citations) AnthropicContentBlock { documentBlock := AnthropicContentBlock{ Type: AnthropicContentBlockTypeDocument, CacheControl: cacheControl, @@ -2263,13 +2284,20 @@ func ConvertResponsesFileBlockToAnthropic(fileBlock *schemas.ResponsesInputMessa documentBlock.Citations = &AnthropicCitations{Config: citations} } - if fileBlock == nil { + // Set title if provided + if fileBlock != nil && fileBlock.Filename != nil { + documentBlock.Title = fileBlock.Filename + } + + // Handle file_id reference + if fileID != nil && *fileID != "" { + documentBlock.Source.SourceObj.Type = "file" + documentBlock.Source.SourceObj.FileID = fileID return documentBlock } - // Set title if provided - if fileBlock.Filename != nil { - documentBlock.Title = fileBlock.Filename + if fileBlock == nil { + return documentBlock } // Handle file_data (base64 encoded data or plain text) diff --git a/core/providers/anthropic/utils_test.go b/core/providers/anthropic/utils_test.go index 8161a9805e9..9f0f3e4d65c 100644 --- a/core/providers/anthropic/utils_test.go +++ b/core/providers/anthropic/utils_test.go @@ -2172,7 +2172,10 @@ func TestGetRequestBodyForResponses_RawBodyStripsFallbacks(t *testing.T) { RawRequestBody: rawBody, } - result, bifrostErr := getRequestBodyForResponses(ctx, request, false, nil, false, false) + result, bifrostErr := BuildAnthropicResponsesRequestBody(ctx, request, AnthropicRequestBuildConfig{ + Provider: schemas.Anthropic, + IsStreaming: false, + }) if bifrostErr != nil { t.Fatalf("unexpected error: %v", bifrostErr) } diff --git a/core/providers/anthropic/websearch_test.go b/core/providers/anthropic/websearch_test.go index 6511af88766..159b415bd33 100644 --- a/core/providers/anthropic/websearch_test.go +++ b/core/providers/anthropic/websearch_test.go @@ -269,3 +269,39 @@ func TestWebSearch_FullFlow_AnyUserAgent(t *testing.T) { t.Errorf("reconstructed query = %v, want %q", got["query"], "latest AI news") } } + +// TestServerSearchTools_VersionRecognition guards the request converter against a +// newer version-dated web_search / web_fetch tool type being silently downgraded to +// a client function tool. convertAnthropicToolToBifrost matches these by prefix +// (mirroring the unmarshaler and the chat path), so any current or future dated +// version must map to the neutral server-tool type. Anti-vacuous: the future-dated +// entries (…20260318) fall through to ResponsesToolTypeFunction before the fix. +func TestServerSearchTools_VersionRecognition(t *testing.T) { + t.Parallel() + cases := []struct { + toolType string + want schemas.ResponsesToolType + }{ + // web_search: known versions + a future-dated one. + {"web_search_20250305", schemas.ResponsesToolTypeWebSearch}, + {"web_search_20260209", schemas.ResponsesToolTypeWebSearch}, + {"web_search_20260318", schemas.ResponsesToolTypeWebSearch}, + // web_fetch: known versions + the reported 20260318. + {"web_fetch_20250910", schemas.ResponsesToolTypeWebFetch}, + {"web_fetch_20260209", schemas.ResponsesToolTypeWebFetch}, + {"web_fetch_20260309", schemas.ResponsesToolTypeWebFetch}, + {"web_fetch_20260318", schemas.ResponsesToolTypeWebFetch}, + } + for _, c := range cases { + t.Run(c.toolType, func(t *testing.T) { + in := &AnthropicTool{Type: schemas.Ptr(AnthropicToolType(c.toolType)), Name: "web_fetch"} + neutral := convertAnthropicToolToBifrost(in) + if neutral == nil { + t.Fatalf("%s: convertAnthropicToolToBifrost returned nil", c.toolType) + } + if neutral.Type != c.want { + t.Errorf("%s: neutral tool type = %q, want %q (must not fall through to a client function tool)", c.toolType, neutral.Type, c.want) + } + }) + } +} diff --git a/core/providers/azure/azure.go b/core/providers/azure/azure.go index 04822817010..af41bcda845 100644 --- a/core/providers/azure/azure.go +++ b/core/providers/azure/azure.go @@ -204,105 +204,6 @@ func (provider *AzureProvider) GetProviderKey() schemas.ModelProvider { return schemas.Azure } -// completeRequest sends a request to Azure's API and handles the response. -// It constructs the API URL, sets up authentication, and processes the response. -// Returns the response body, request latency, or an error if the request fails. -func (provider *AzureProvider) completeRequest( - ctx *schemas.BifrostContext, - jsonData []byte, - path string, - key schemas.Key, - model string, -) ([]byte, time.Duration, map[string]string, *schemas.BifrostError) { - // Create the request with the JSON body - req := fasthttp.AcquireRequest() - resp := fasthttp.AcquireResponse() - defer fasthttp.ReleaseRequest(req) - respOwned := true - defer func() { - if respOwned { - fasthttp.ReleaseResponse(resp) - } - }() - - var url string - - // Set any extra headers from network config. - // For Anthropic models, exclude anthropic-beta — it is merged and filtered explicitly below. - if schemas.IsAnthropicModelFamily(ctx, model) { - providerUtils.SetExtraHeaders(ctx, req, provider.networkConfig.ExtraHeaders, []string{anthropic.AnthropicBetaHeader}) - } else { - providerUtils.SetExtraHeaders(ctx, req, provider.networkConfig.ExtraHeaders, nil) - } - req.Header.SetMethod(http.MethodPost) - req.Header.SetContentType("application/json") - - // Get authentication headers - authHeaders, bifrostErr := provider.getAzureAuthHeaders(ctx, key, schemas.IsAnthropicModelFamily(ctx, model)) - if bifrostErr != nil { - return nil, 0, nil, bifrostErr - } - - // Apply headers to request - for k, v := range authHeaders { - req.Header.Set(k, v) - } - - endpoint := resolveAzureEndpoint(ctx, key) - if endpoint == "" { - return nil, 0, nil, providerUtils.NewConfigurationError("endpoint not set") - } - - if schemas.IsAnthropicModelFamily(ctx, model) { - req.Header.Set("anthropic-version", resolveAnthropicVersion(ctx)) - url = fmt.Sprintf("%s/%s", endpoint, path) - - // Merge ExtraHeaders + context anthropic-beta, filter for Azure, then set as HTTP header - if betaHeaders := anthropic.FilterBetaHeadersForProvider(anthropic.MergeBetaHeaders(ctx, provider.networkConfig.ExtraHeaders), schemas.Azure, provider.networkConfig.BetaHeaderOverrides); len(betaHeaders) > 0 { - req.Header.Set(anthropic.AnthropicBetaHeader, strings.Join(betaHeaders, ",")) - } else { - req.Header.Del(anthropic.AnthropicBetaHeader) - } - } else { - url = fmt.Sprintf("%s/%s", endpoint, path) - } - - req.SetRequestURI(url) - if !providerUtils.ApplyLargePayloadRequestBodyWithModelNormalization(ctx, req, schemas.OpenAI) { - req.SetBody(jsonData) - } - - // Send the request with optional large response streaming - activeClient := providerUtils.PrepareResponseStreaming(ctx, provider.client, resp) - latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) - defer wait() - if bifrostErr != nil { - return nil, latency, nil, bifrostErr - } - - // Extract provider response headers before body is copied — do this before status check - // so error responses also carry provider headers (rate-limit info, request IDs, etc.) - providerResponseHeaders := providerUtils.ExtractProviderResponseHeaders(resp) - - // Handle error response - if resp.StatusCode() != fasthttp.StatusOK { - providerUtils.MaterializeStreamErrorBody(ctx, resp) - rawErrBody := append([]byte(nil), resp.Body()...) - return rawErrBody, latency, providerResponseHeaders, openai.ParseOpenAIError(resp) - } - - body, isLargeResp, decodeErr := providerUtils.FinalizeResponseWithLargeDetection(ctx, resp, provider.logger) - if decodeErr != nil { - return nil, latency, providerResponseHeaders, decodeErr - } - if isLargeResp { - respOwned = false - return nil, latency, providerResponseHeaders, nil - } - - return body, latency, providerResponseHeaders, nil -} - // listModelsByKey performs a list models request for a single key. // Returns the response and latency, or an error if the request fails. func (provider *AzureProvider) listModelsByKey(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostListModelsRequest) (*schemas.BifrostListModelsResponse, *schemas.BifrostError) { @@ -345,7 +246,7 @@ func (provider *AzureProvider) listModelsByKey(ctx *schemas.BifrostContext, key // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, openai.ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(openai.ParseOpenAIError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -401,63 +302,28 @@ func (provider *AzureProvider) ListModels(ctx *schemas.BifrostContext, keys []sc // It formats the request, sends it to Azure, and processes the response. // Returns a BifrostResponse containing the completion results or an error if the request fails. func (provider *AzureProvider) TextCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostTextCompletionRequest) (*schemas.BifrostTextCompletionResponse, *schemas.BifrostError) { - // Use centralized OpenAI text converter (Azure is OpenAI-compatible) - jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody( - ctx, - request, - func() (providerUtils.RequestBodyWithExtraParams, error) { - return openai.ToOpenAITextCompletionRequest(request), nil - }) + endpoint := resolveAzureEndpoint(ctx, key) + if endpoint == "" { + return nil, providerUtils.NewConfigurationError("endpoint not set") + } + authHeader, bifrostErr := provider.getAzureAuthHeaders(ctx, key, false) if bifrostErr != nil { return nil, bifrostErr } - - responseBody, latency, providerResponseHeaders, err := provider.completeRequest( + return openai.HandleOpenAITextCompletionRequest( ctx, - jsonData, - "openai/v1/completions", - key, - request.Model, + provider.client, + fmt.Sprintf("%s/openai/v1/completions", endpoint), + request, + authHeader, + provider.networkConfig.ExtraHeaders, + provider.GetProviderKey(), + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + nil, + nil, + provider.logger, ) - if providerResponseHeaders != nil { - ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) - } - if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) - } - - // Large response mode: return lightweight response with metadata only - if isLargeResp, _ := ctx.Value(schemas.BifrostContextKeyLargeResponseMode).(bool); isLargeResp { - return &schemas.BifrostTextCompletionResponse{ - Model: request.Model, - ExtraFields: schemas.BifrostResponseExtraFields{ - Latency: latency.Milliseconds(), - ProviderResponseHeaders: providerResponseHeaders, - }, - }, nil - } - - response := &schemas.BifrostTextCompletionResponse{} - - rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, response, jsonData, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) - if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) - } - - response.ExtraFields.Latency = latency.Milliseconds() - response.ExtraFields.ProviderResponseHeaders = providerResponseHeaders - - // Set raw request if enabled - if providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest) { - response.ExtraFields.RawRequest = rawRequest - } - - // Set raw response if enabled - if providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) { - response.ExtraFields.RawResponse = rawResponse - } - - return response, nil } // TextCompletionStream performs a streaming text completion request to Azure's API. @@ -500,93 +366,58 @@ func (provider *AzureProvider) TextCompletionStream(ctx *schemas.BifrostContext, // It formats the request, sends it to Azure, and processes the response. // Returns a BifrostResponse containing the completion results or an error if the request fails. func (provider *AzureProvider) ChatCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostChatRequest) (*schemas.BifrostChatResponse, *schemas.BifrostError) { - jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody( - ctx, - request, - func() (providerUtils.RequestBodyWithExtraParams, error) { - if schemas.ResolveFamily(ctx, request.Model) == schemas.ModelFamilyAnthropic { - reqBody, err := anthropic.ToAnthropicChatRequest(ctx, request) - if err != nil { - return nil, err - } - if reqBody != nil { - // Add provider-aware beta headers for Azure - anthropic.AddMissingBetaHeadersToContext(ctx, reqBody, schemas.Azure) - } - return reqBody, nil - } else { - return openai.ToOpenAIChatRequest(ctx, request), nil - } - }) - if bifrostErr != nil { - return nil, bifrostErr - } - - var path string - if schemas.ResolveFamily(ctx, request.Model) == schemas.ModelFamilyAnthropic { - path = "anthropic/v1/messages" - } else { - path = "openai/v1/chat/completions" - } - - responseBody, latency, providerResponseHeaders, err := provider.completeRequest( - ctx, - jsonData, - path, - key, - request.Model, - ) - if providerResponseHeaders != nil { - ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) - } - if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) - } - - // Large response mode: return lightweight response with metadata only - if isLargeResp, _ := ctx.Value(schemas.BifrostContextKeyLargeResponseMode).(bool); isLargeResp { - return &schemas.BifrostChatResponse{ - Model: request.Model, - ExtraFields: schemas.BifrostResponseExtraFields{ - Latency: latency.Milliseconds(), - ProviderResponseHeaders: providerResponseHeaders, - }, - }, nil + endpoint := resolveAzureEndpoint(ctx, key) + if endpoint == "" { + return nil, providerUtils.NewConfigurationError("endpoint not set") } - response := &schemas.BifrostChatResponse{} - var rawRequest interface{} - var rawResponse interface{} - - if schemas.ResolveFamily(ctx, request.Model) == schemas.ModelFamilyAnthropic { - anthropicResponse := anthropic.AcquireAnthropicMessageResponse() - defer anthropic.ReleaseAnthropicMessageResponse(anthropicResponse) - rawRequest, rawResponse, bifrostErr = providerUtils.HandleProviderResponse(responseBody, anthropicResponse, jsonData, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) - if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) - } - response = anthropicResponse.ToBifrostChatResponse(ctx) - } else { - rawRequest, rawResponse, bifrostErr = providerUtils.HandleProviderResponse(responseBody, response, jsonData, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) + if schemas.IsAnthropicModelFamily(ctx, request.Model) { + // Anthropic-family models use the native Anthropic Messages endpoint via the shared handler. + authHeader, bifrostErr := provider.getAzureAuthHeaders(ctx, key, true) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, bifrostErr } + authHeader["anthropic-version"] = resolveAnthropicVersion(ctx) + return anthropic.HandleAnthropicChatCompletionRequest( + ctx, + provider.client, + fmt.Sprintf("%s/anthropic/v1/messages", endpoint), + request, + anthropic.AnthropicRequestBuildConfig{ + Provider: schemas.Azure, + Model: request.Model, + IsStreaming: false, + BetaHeaderOverrides: provider.networkConfig.BetaHeaderOverrides, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }, + authHeader, + provider.networkConfig.ExtraHeaders, + nil, + provider.logger, + ) } - response.ExtraFields.Latency = latency.Milliseconds() - response.ExtraFields.ProviderResponseHeaders = providerResponseHeaders - - // Set raw request if enabled - if providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest) { - response.ExtraFields.RawRequest = rawRequest - } - - // Set raw response if enabled - if providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) { - response.ExtraFields.RawResponse = rawResponse + // OpenAI-family models use the OpenAI-compatible Azure endpoint via the shared handler. + authHeader, bifrostErr := provider.getAzureAuthHeaders(ctx, key, false) + if bifrostErr != nil { + return nil, bifrostErr } - - return response, nil + return openai.HandleOpenAIChatCompletionRequest( + ctx, + provider.client, + fmt.Sprintf("%s/openai/v1/chat/completions", endpoint), + request, + authHeader, + provider.networkConfig.ExtraHeaders, + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + provider.GetProviderKey(), + nil, + nil, + nil, + provider.logger, + ) } // ChatCompletionStream performs a streaming chat completion request to Azure's API. @@ -607,21 +438,13 @@ func (provider *AzureProvider) ChatCompletionStream(ctx *schemas.BifrostContext, authHeader["anthropic-version"] = resolveAnthropicVersion(ctx) url = fmt.Sprintf("%s/anthropic/v1/messages", endpoint) - jsonData, err := providerUtils.CheckContextAndGetRequestBody( - ctx, - request, - func() (providerUtils.RequestBodyWithExtraParams, error) { - reqBody, err := anthropic.ToAnthropicChatRequest(ctx, request) - if err != nil { - return nil, err - } - if reqBody != nil { - reqBody.Stream = schemas.Ptr(true) - // Add provider-aware beta headers for Azure - anthropic.AddMissingBetaHeadersToContext(ctx, reqBody, schemas.Azure) - } - return reqBody, nil - }) + jsonData, err := anthropic.BuildAnthropicChatRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{ + Provider: schemas.Azure, + Model: request.Model, + IsStreaming: true, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }) if err != nil { return nil, err } @@ -641,6 +464,7 @@ func (provider *AzureProvider) ChatCompletionStream(ctx *schemas.BifrostContext, provider.GetProviderKey(), postHookRunner, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -669,6 +493,7 @@ func (provider *AzureProvider) ChatCompletionStream(ctx *schemas.BifrostContext, nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -679,88 +504,60 @@ func (provider *AzureProvider) ChatCompletionStream(ctx *schemas.BifrostContext, // It formats the request, sends it to Azure, and processes the response. // Returns a BifrostResponse containing the completion results or an error if the request fails. func (provider *AzureProvider) Responses(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostResponsesRequest) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) { - var jsonData []byte - var bifrostErr *schemas.BifrostError + endpoint := resolveAzureEndpoint(ctx, key) + if endpoint == "" { + return nil, providerUtils.NewConfigurationError("endpoint not set") + } + if schemas.IsAnthropicModelFamily(ctx, request.Model) { - jsonData, bifrostErr = getRequestBodyForAnthropicResponses(ctx, request, request.Model, false, provider.sendBackRawRequest, provider.sendBackRawResponse) - } else { - jsonData, bifrostErr = providerUtils.CheckContextAndGetRequestBody( + // Anthropic-family models use the native Anthropic Messages endpoint via the shared handler. + authHeader, bifrostErr := provider.getAzureAuthHeaders(ctx, key, true) + if bifrostErr != nil { + return nil, bifrostErr + } + authHeader["anthropic-version"] = resolveAnthropicVersion(ctx) + return anthropic.HandleAnthropicResponsesRequest( ctx, + provider.client, + fmt.Sprintf("%s/anthropic/v1/messages", endpoint), request, - func() (providerUtils.RequestBodyWithExtraParams, error) { - reqBody := openai.ToOpenAIResponsesRequest(ctx, request) - return reqBody, nil - }) + anthropic.AnthropicRequestBuildConfig{ + Provider: schemas.Azure, + Model: request.Model, + IsStreaming: false, + ValidateTools: true, + BetaHeaderOverrides: provider.networkConfig.BetaHeaderOverrides, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }, + authHeader, + provider.networkConfig.ExtraHeaders, + nil, + provider.logger, + ) } + + // OpenAI-family models use the OpenAI-compatible Azure endpoint via the shared handler. + authHeader, bifrostErr := provider.getAzureAuthHeaders(ctx, key, false) if bifrostErr != nil { return nil, bifrostErr } - - var path string - if schemas.IsAnthropicModelFamily(ctx, request.Model) { - path = "anthropic/v1/messages" - } else { - path = fmt.Sprintf("openai/v1/responses?api-version=%s", resolveAPIVersion(ctx, AzureAPIVersionPreview)) - } - - responseBody, latency, providerResponseHeaders, err := provider.completeRequest( + path := fmt.Sprintf("openai/v1/responses?api-version=%s", resolveAPIVersion(ctx, AzureAPIVersionPreview)) + return openai.HandleOpenAIResponsesRequest( ctx, - jsonData, - path, - key, - request.Model, + provider.client, + fmt.Sprintf("%s/%s", endpoint, path), + request, + authHeader, + provider.networkConfig.ExtraHeaders, + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + provider.GetProviderKey(), + nil, + nil, + nil, + provider.logger, ) - if providerResponseHeaders != nil { - ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) - } - if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) - } - - // Large response mode: return lightweight response with metadata only - if isLargeResp, _ := ctx.Value(schemas.BifrostContextKeyLargeResponseMode).(bool); isLargeResp { - return &schemas.BifrostResponsesResponse{ - Model: request.Model, - ExtraFields: schemas.BifrostResponseExtraFields{ - Latency: latency.Milliseconds(), - ProviderResponseHeaders: providerResponseHeaders, - }, - }, nil - } - - response := &schemas.BifrostResponsesResponse{} - var rawRequest interface{} - var rawResponse interface{} - - if schemas.IsAnthropicModelFamily(ctx, request.Model) { - anthropicResponse := anthropic.AcquireAnthropicMessageResponse() - defer anthropic.ReleaseAnthropicMessageResponse(anthropicResponse) - rawRequest, rawResponse, bifrostErr = providerUtils.HandleProviderResponse(responseBody, anthropicResponse, jsonData, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) - if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) - } - response = anthropicResponse.ToBifrostResponsesResponse(ctx) - } else { - rawRequest, rawResponse, bifrostErr = providerUtils.HandleProviderResponse(responseBody, response, jsonData, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) - if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) - } - } - - response.ExtraFields.Latency = latency.Milliseconds() - response.ExtraFields.ProviderResponseHeaders = providerResponseHeaders - - // Set raw request if enabled - if providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest) { - response.ExtraFields.RawRequest = rawRequest - } - - // Set raw response if enabled - if providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) { - response.ExtraFields.RawResponse = rawResponse - } - - return response, nil } // ResponsesStream performs a streaming responses request to Azure's API. @@ -778,7 +575,14 @@ func (provider *AzureProvider) ResponsesStream(ctx *schemas.BifrostContext, post authHeader["anthropic-version"] = resolveAnthropicVersion(ctx) url = fmt.Sprintf("%s/anthropic/v1/messages", endpoint) - jsonData, bifrostErr := getRequestBodyForAnthropicResponses(ctx, request, request.Model, true, provider.sendBackRawRequest, provider.sendBackRawResponse) + jsonData, bifrostErr := anthropic.BuildAnthropicResponsesRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{ + Provider: schemas.Azure, + Model: request.Model, + IsStreaming: true, + ValidateTools: true, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }) if bifrostErr != nil { return nil, bifrostErr } @@ -798,6 +602,7 @@ func (provider *AzureProvider) ResponsesStream(ctx *schemas.BifrostContext, post provider.GetProviderKey(), postHookRunner, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -825,6 +630,7 @@ func (provider *AzureProvider) ResponsesStream(ctx *schemas.BifrostContext, post nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -835,64 +641,27 @@ func (provider *AzureProvider) ResponsesStream(ctx *schemas.BifrostContext, post // The input can be either a single string or a slice of strings for batch embedding. // Returns a BifrostResponse containing the embedding(s) and any error that occurred. func (provider *AzureProvider) Embedding(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostEmbeddingRequest) (*schemas.BifrostEmbeddingResponse, *schemas.BifrostError) { - // Use centralized converter - jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody( - ctx, - request, - func() (providerUtils.RequestBodyWithExtraParams, error) { - return openai.ToOpenAIEmbeddingRequest(request), nil - }) + endpoint := resolveAzureEndpoint(ctx, key) + if endpoint == "" { + return nil, providerUtils.NewConfigurationError("endpoint not set") + } + authHeader, bifrostErr := provider.getAzureAuthHeaders(ctx, key, false) if bifrostErr != nil { return nil, bifrostErr } - - responseBody, latency, providerResponseHeaders, err := provider.completeRequest( + return openai.HandleOpenAIEmbeddingRequest( ctx, - jsonData, - "openai/v1/embeddings", - key, - request.Model, + provider.client, + fmt.Sprintf("%s/openai/v1/embeddings", endpoint), + request, + authHeader, + provider.networkConfig.ExtraHeaders, + provider.GetProviderKey(), + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + nil, + provider.logger, ) - if providerResponseHeaders != nil { - ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) - } - if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) - } - - // Large response mode: return lightweight response with metadata only - if isLargeResp, _ := ctx.Value(schemas.BifrostContextKeyLargeResponseMode).(bool); isLargeResp { - return &schemas.BifrostEmbeddingResponse{ - Model: request.Model, - ExtraFields: schemas.BifrostResponseExtraFields{ - Latency: latency.Milliseconds(), - ProviderResponseHeaders: providerResponseHeaders, - }, - }, nil - } - - response := &schemas.BifrostEmbeddingResponse{} - - // Use enhanced response handler with pre-allocated response - rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, response, jsonData, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) - if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) - } - - response.ExtraFields.Latency = latency.Milliseconds() - response.ExtraFields.ProviderResponseHeaders = providerResponseHeaders - - // Set raw request if enabled - if providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest) { - response.ExtraFields.RawRequest = rawRequest - } - - // Set raw response if enabled - if providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) { - response.ExtraFields.RawResponse = rawResponse - } - - return response, nil } // Speech is not supported by the Azure provider. @@ -1003,6 +772,7 @@ func (provider *AzureProvider) SpeechStream(ctx *schemas.BifrostContext, postHoo startTime := time.Now() // Make the request requestErr := provider.client.Do(req, resp) + latency := time.Since(startTime) if requestErr != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(requestErr, context.Canceled) { @@ -1013,16 +783,16 @@ func (provider *AzureProvider) SpeechStream(ctx *schemas.BifrostContext, postHoo Message: schemas.ErrRequestCancelled, Error: requestErr, }, - }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } if errors.Is(requestErr, fasthttp.ErrTimeout) || errors.Is(requestErr, context.DeadlineExceeded) { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, requestErr), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, requestErr), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Request failed before the first response byte (server closed an idle/pooled connection, // broken pipe, connection refused, DNS failure, etc.). Surface as a retriable upstream // connection error (502) so executeRequestWithRetries honors max_retries, matching the // non-streaming path - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, requestErr), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, requestErr), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Extract provider response headers before status check so error responses also forward them @@ -1031,7 +801,7 @@ func (provider *AzureProvider) SpeechStream(ctx *schemas.BifrostContext, postHoo // Check for HTTP errors if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) - return nil, providerUtils.EnrichError(ctx, openai.ParseOpenAIError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, openai.ParseOpenAIError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Create response channel @@ -1517,7 +1287,7 @@ func (provider *AzureProvider) VideoDownload(ctx *schemas.BifrostContext, key sc // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, openai.ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(openai.ParseOpenAIError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -1686,7 +1456,7 @@ func (provider *AzureProvider) FileUpload(ctx *schemas.BifrostContext, key schem // Handle error response if resp.StatusCode() != fasthttp.StatusOK && resp.StatusCode() != fasthttp.StatusCreated { - return nil, openai.ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(openai.ParseOpenAIError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -1778,7 +1548,7 @@ func (provider *AzureProvider) FileList(ctx *schemas.BifrostContext, keys []sche // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, openai.ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(openai.ParseOpenAIError(resp), latency) } body, decodeErr := providerUtils.CheckAndDecodeBody(resp) @@ -2214,17 +1984,17 @@ func (provider *AzureProvider) BatchCreate(ctx *schemas.BifrostContext, key sche latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Handle error response if resp.StatusCode() != fasthttp.StatusOK && resp.StatusCode() != fasthttp.StatusCreated { - return nil, providerUtils.EnrichError(ctx, openai.ParseOpenAIError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, openai.ParseOpenAIError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } var openAIResp openai.OpenAIBatchResponse @@ -2232,7 +2002,7 @@ func (provider *AzureProvider) BatchCreate(ctx *schemas.BifrostContext, key sche sendBackRawResponse := providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(body, &openAIResp, jsonData, sendBackRawRequest, sendBackRawResponse) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } return openAIResp.ToBifrostBatchCreateResponse(latency, sendBackRawRequest, sendBackRawResponse, rawRequest, rawResponse), nil @@ -2309,7 +2079,7 @@ func (provider *AzureProvider) BatchList(ctx *schemas.BifrostContext, keys []sch // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, openai.ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(openai.ParseOpenAIError(resp), latency) } body, decodeErr := providerUtils.CheckAndDecodeBody(resp) @@ -2751,52 +2521,27 @@ func (provider *AzureProvider) CountTokens(_ *schemas.BifrostContext, _ schemas. // Compaction compacts a conversation context window using Azure OpenAI's /openai/v1/responses/compact endpoint. func (provider *AzureProvider) Compaction(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostCompactionRequest) (*schemas.BifrostCompactionResponse, *schemas.BifrostError) { - jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody( - ctx, - request, - func() (providerUtils.RequestBodyWithExtraParams, error) { - return openai.ToOpenAICompactionRequest(ctx, request), nil - }) - if bifrostErr != nil { - return nil, bifrostErr - } - - path := fmt.Sprintf("openai/v1/responses/compact?api-version=%s", AzureAPIVersionPreview) - - responseBody, latency, providerResponseHeaders, err := provider.completeRequest(ctx, jsonData, path, key, request.Model) - if providerResponseHeaders != nil { - ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) - } - if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) - } - - if isLargeResp, _ := ctx.Value(schemas.BifrostContextKeyLargeResponseMode).(bool); isLargeResp { - return &schemas.BifrostCompactionResponse{ - ExtraFields: schemas.BifrostResponseExtraFields{ - Latency: latency.Milliseconds(), - ProviderResponseHeaders: providerResponseHeaders, - }, - }, nil + endpoint := resolveAzureEndpoint(ctx, key) + if endpoint == "" { + return nil, providerUtils.NewConfigurationError("endpoint not set") } - - response := &schemas.BifrostCompactionResponse{} - rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, response, jsonData, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) + authHeader, bifrostErr := provider.getAzureAuthHeaders(ctx, key, false) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) - } - - response.ExtraFields.Latency = latency.Milliseconds() - response.ExtraFields.ProviderResponseHeaders = providerResponseHeaders - - if providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest) { - response.ExtraFields.RawRequest = rawRequest - } - if providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) { - response.ExtraFields.RawResponse = rawResponse + return nil, bifrostErr } - - return response, nil + path := fmt.Sprintf("openai/v1/responses/compact?api-version=%s", resolveAPIVersion(ctx, AzureAPIVersionPreview)) + return openai.HandleOpenAICompactionRequest( + ctx, + provider.client, + fmt.Sprintf("%s/%s", endpoint, path), + request, + authHeader, + provider.networkConfig.ExtraHeaders, + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + provider.GetProviderKey(), + provider.logger, + ) } // buildContainerURL constructs the Azure container API URL. @@ -2871,7 +2616,7 @@ func (provider *AzureProvider) ContainerCreate(ctx *schemas.BifrostContext, key return nil, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK && resp.StatusCode() != fasthttp.StatusCreated { - return nil, openai.ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(openai.ParseOpenAIError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -3176,7 +2921,7 @@ func (provider *AzureProvider) ContainerFileCreate(ctx *schemas.BifrostContext, return nil, bifrostErr } if resp.StatusCode() >= 400 { - return nil, openai.ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(openai.ParseOpenAIError(resp), latency) } responseBody, err := providerUtils.CheckAndDecodeBody(resp) @@ -3284,7 +3029,7 @@ func (provider *AzureProvider) ContainerFileList(ctx *schemas.BifrostContext, ke return nil, bifrostErr } if resp.StatusCode() >= 400 { - return nil, openai.ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(openai.ParseOpenAIError(resp), latency) } responseBody, err := providerUtils.CheckAndDecodeBody(resp) @@ -3737,26 +3482,28 @@ func (provider *AzureProvider) PassthroughStream( startTime := time.Now() - if err := activeClient.Do(fasthttpReq, resp); err != nil { + err = activeClient.Do(fasthttpReq, resp) + latency := time.Since(startTime) + if err != nil { providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } // Request failed before the first response byte (server closed an idle/pooled connection, // broken pipe, connection refused, DNS failure, etc.). Surface as a retriable upstream // connection error (502) so executeRequestWithRetries honors max_retries, matching the // non-streaming path - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), latency) } headers := providerUtils.ExtractPassthroughProviderResponseHeaders(resp) diff --git a/core/providers/azure/realtime.go b/core/providers/azure/realtime.go index 887e92ef6b0..8ef2fb5ba08 100644 --- a/core/providers/azure/realtime.go +++ b/core/providers/azure/realtime.go @@ -121,7 +121,7 @@ func (provider *AzureProvider) ExchangeRealtimeWebRTCSDP( } req.SetBody(bodyBuf.Bytes()) - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { return "", bifrostErr @@ -129,7 +129,7 @@ func (provider *AzureProvider) ExchangeRealtimeWebRTCSDP( answerBody := resp.Body() if resp.StatusCode() < fasthttp.StatusOK || resp.StatusCode() >= fasthttp.StatusMultipleChoices { - return "", provider.realtimeWebRTCUpstreamError(ctx, resp.StatusCode(), answerBody) + return "", providerUtils.SetErrorLatency(provider.realtimeWebRTCUpstreamError(ctx, resp.StatusCode(), answerBody), latency) } return string(answerBody), nil @@ -252,7 +252,7 @@ func (provider *AzureProvider) CreateRealtimeClientSecret( ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, headers) if resp.StatusCode() < fasthttp.StatusOK || resp.StatusCode() >= fasthttp.StatusMultipleChoices { - return nil, provider.parseRealtimeClientSecretError(ctx, resp) + return nil, providerUtils.SetErrorLatency(provider.parseRealtimeClientSecretError(ctx, resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -331,7 +331,6 @@ func newAzureRealtimeError(status int, errorType, message string, err error) *sc return bifrostErr } - func (provider *AzureProvider) parseRealtimeClientSecretError(ctx *schemas.BifrostContext, resp *fasthttp.Response) *schemas.BifrostError { body, _ := providerUtils.CheckAndDecodeBody(resp) var parsed struct { diff --git a/core/providers/azure/utils.go b/core/providers/azure/utils.go index 2aecf5e04f6..1fd34e79a71 100644 --- a/core/providers/azure/utils.go +++ b/core/providers/azure/utils.go @@ -3,23 +3,9 @@ package azure import ( "strings" - "github.com/maximhq/bifrost/core/providers/anthropic" - "github.com/maximhq/bifrost/core/schemas" + schemas "github.com/maximhq/bifrost/core/schemas" ) -// getRequestBodyForAnthropicResponses serializes a BifrostResponsesRequest into the Anthropic wire format for Azure. -// It delegates to BuildAnthropicResponsesRequestBody with the Azure provider and the target deployment name. -func getRequestBodyForAnthropicResponses(ctx *schemas.BifrostContext, request *schemas.BifrostResponsesRequest, deployment string, isStreaming bool, shouldSendBackRawRequest bool, shouldSendBackRawResponse bool) ([]byte, *schemas.BifrostError) { - return anthropic.BuildAnthropicResponsesRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{ - Provider: schemas.Azure, - Deployment: deployment, - IsStreaming: isStreaming, - ValidateTools: true, - ShouldSendBackRawRequest: shouldSendBackRawRequest, - ShouldSendBackRawResponse: shouldSendBackRawResponse, - }) -} - // getAzureScopes returns the configured scopes or the default scope if none are valid. // It filters out empty/whitespace-only strings. func getAzureScopes(configuredScopes []string) []string { diff --git a/core/providers/bedrock/bedrock.go b/core/providers/bedrock/bedrock.go index 37b5187174d..094d920fbe0 100644 --- a/core/providers/bedrock/bedrock.go +++ b/core/providers/bedrock/bedrock.go @@ -37,7 +37,7 @@ type BedrockProvider struct { logger schemas.Logger // Logger for provider operations client *http.Client // HTTP client for unary API requests (Client.Timeout bounds overall response) streamingClient *http.Client // HTTP client for streaming API requests (no Timeout; idle governed by NewIdleTimeoutReader) - mantleClient *fasthttp.Client // fasthttp client for Bedrock Mantle (OpenAI-compatible) requests + mantleClient *fasthttp.Client // fasthttp client for Bedrock Mantle unary requests (OpenAI-compatible and native-Anthropic paths) mantleStreamingClient *fasthttp.Client // fasthttp streaming client for Bedrock Mantle streaming requests networkConfig schemas.NetworkConfig // Network configuration including extra headers customProviderConfig *schemas.CustomProviderConfig // Custom provider config @@ -125,7 +125,9 @@ func NewBedrockProvider(config *schemas.ProviderConfig, logger schemas.Logger) ( client := &http.Client{Transport: transport, Timeout: requestTimeout} streamingClient := providerUtils.BuildStreamingHTTPClient(client) - // fasthttp clients for Bedrock Mantle (OpenAI-compatible endpoint) + // fasthttp clients for Bedrock Mantle (shared by OpenAI-compatible and native-Anthropic paths). + // ReadTimeout is the shared provider request timeout, not an OpenAI-specific value; oversized + // Anthropic responses are handled by PrepareResponseStreaming, not by these static settings. mantleFasthttpClient := &fasthttp.Client{ ReadTimeout: requestTimeout, WriteTimeout: requestTimeout, @@ -293,53 +295,61 @@ func (provider *BedrockProvider) completeRequest(ctx *schemas.BifrostContext, js req.Header.Set("Authorization", fmt.Sprintf("Bearer %s", key.Value.GetValue())) } else { // Sign the request using either explicit credentials or IAM role authentication - if err := signAWSRequest(ctx, req, config.AccessKey, config.SecretKey, config.SessionToken, config.RoleARN, config.ExternalID, config.RoleSessionName, region, bedrockSigningService); err != nil { + if err := signAWSRequest(ctx, req, config, region, bedrockSigningService); err != nil { return nil, 0, nil, err } } + body, latency, providerResponseHeaders, bErr := provider.executeBedrockRequest(req) + return body, latency, providerResponseHeaders, bErr +} + +// executeBedrockRequest sends an already-built (and authenticated) request via the +// unary HTTP client, measures latency, and parses a Bedrock error envelope on non-200 +// responses. Used by completeRequest for the bedrock-runtime (Converse) path. +func (provider *BedrockProvider) executeBedrockRequest(req *http.Request) ([]byte, time.Duration, map[string]string, *schemas.BifrostError) { // Execute the request and measure latency startTime := time.Now() resp, err := provider.client.Do(req) latency := time.Since(startTime) if err != nil { if errors.Is(err, context.Canceled) { - return nil, latency, nil, &schemas.BifrostError{ + return nil, latency, nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } // Check for timeout first using net.Error before checking net.OpError var netErr net.Error if errors.As(err, &netErr) && netErr.Timeout() { - return nil, latency, nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, latency, nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } if errors.Is(err, http.ErrHandlerTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, latency, nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, latency, nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } // Check for DNS lookup and network errors after timeout checks var opErr *net.OpError var dnsErr *net.DNSError if errors.As(err, &opErr) || errors.As(err, &dnsErr) { - return nil, latency, nil, &schemas.BifrostError{ + return nil, latency, nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Message: schemas.ErrProviderNetworkError, Error: err, }, - } + }, latency) } - return nil, latency, nil, &schemas.BifrostError{ + return nil, latency, nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Message: schemas.ErrProviderDoRequest, Error: err, }, - } + }, latency) } // Extract provider response headers before closing the body @@ -349,46 +359,17 @@ func (provider *BedrockProvider) completeRequest(ctx *schemas.BifrostContext, js // Read response body body, err := io.ReadAll(resp.Body) if err != nil { - return nil, latency, providerResponseHeaders, &schemas.BifrostError{ + return nil, latency, providerResponseHeaders, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: true, Error: &schemas.ErrorField{ Message: "error reading request", Error: err, }, - } + }, latency) } if resp.StatusCode != http.StatusOK { - var errorResp BedrockError - - var rawErrorResponse interface{} - if err := sonic.Unmarshal(body, &rawErrorResponse); err != nil { - rawErrorResponse = string(body) - } - - if err := sonic.Unmarshal(body, &errorResp); err != nil { - return nil, latency, providerResponseHeaders, &schemas.BifrostError{ - IsBifrostError: true, - StatusCode: &resp.StatusCode, - Error: &schemas.ErrorField{ - Message: schemas.ErrProviderResponseUnmarshal, - Error: err, - }, - ExtraFields: schemas.BifrostErrorExtraFields{ - RawResponse: rawErrorResponse, - }, - } - } - - return nil, latency, providerResponseHeaders, &schemas.BifrostError{ - StatusCode: &resp.StatusCode, - Error: &schemas.ErrorField{ - Message: errorResp.Message, - }, - ExtraFields: schemas.BifrostErrorExtraFields{ - RawResponse: rawErrorResponse, - }, - } + return nil, latency, providerResponseHeaders, providerUtils.SetErrorLatency(parseBedrockHTTPError(resp.StatusCode, resp.Header, body), latency) } return body, latency, providerResponseHeaders, nil @@ -420,7 +401,7 @@ func (provider *BedrockProvider) completeAgentRuntimeRequest(ctx *schemas.Bifros if key.Value.GetValue() != "" { req.Header.Set("Authorization", fmt.Sprintf("Bearer %s", key.Value.GetValue())) } else { - if err := signAWSRequest(ctx, req, config.AccessKey, config.SecretKey, config.SessionToken, config.RoleARN, config.ExternalID, config.RoleSessionName, region, bedrockSigningService); err != nil { + if err := signAWSRequest(ctx, req, config, region, bedrockSigningService); err != nil { return nil, 0, nil, err } } @@ -430,40 +411,40 @@ func (provider *BedrockProvider) completeAgentRuntimeRequest(ctx *schemas.Bifros latency := time.Since(startTime) if err != nil { if errors.Is(err, context.Canceled) { - return nil, latency, nil, &schemas.BifrostError{ + return nil, latency, nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } var netErr net.Error if errors.As(err, &netErr) && netErr.Timeout() { - return nil, latency, nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, latency, nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } if errors.Is(err, http.ErrHandlerTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, latency, nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, latency, nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } var opErr *net.OpError var dnsErr *net.DNSError if errors.As(err, &opErr) || errors.As(err, &dnsErr) { - return nil, latency, nil, &schemas.BifrostError{ + return nil, latency, nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Message: schemas.ErrProviderNetworkError, Error: err, }, - } + }, latency) } - return nil, latency, nil, &schemas.BifrostError{ + return nil, latency, nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Message: schemas.ErrProviderDoRequest, Error: err, }, - } + }, latency) } // Extract provider response headers before closing the body @@ -472,17 +453,17 @@ func (provider *BedrockProvider) completeAgentRuntimeRequest(ctx *schemas.Bifros body, err := io.ReadAll(resp.Body) if err != nil { - return nil, latency, providerResponseHeaders, &schemas.BifrostError{ + return nil, latency, providerResponseHeaders, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: true, Error: &schemas.ErrorField{ Message: "error reading request", Error: err, }, - } + }, latency) } if resp.StatusCode != http.StatusOK { - return nil, latency, providerResponseHeaders, parseBedrockHTTPError(resp.StatusCode, resp.Header, body) + return nil, latency, providerResponseHeaders, providerUtils.SetErrorLatency(parseBedrockHTTPError(resp.StatusCode, resp.Header, body), latency) } return body, latency, providerResponseHeaders, nil @@ -521,51 +502,53 @@ func (provider *BedrockProvider) makeStreamingRequest(ctx *schemas.BifrostContex req.Header.Set("Authorization", fmt.Sprintf("Bearer %s", key.Value.GetValue())) } else { // Sign the request using either explicit credentials or IAM role authentication - if err := signAWSRequest(ctx, req, key.BedrockKeyConfig.AccessKey, key.BedrockKeyConfig.SecretKey, key.BedrockKeyConfig.SessionToken, key.BedrockKeyConfig.RoleARN, key.BedrockKeyConfig.ExternalID, key.BedrockKeyConfig.RoleSessionName, region, bedrockSigningService); err != nil { + if err := signAWSRequest(ctx, req, key.BedrockKeyConfig, region, bedrockSigningService); err != nil { return nil, err } } // Make the request + startTime := time.Now() resp, respErr := provider.streamingClient.Do(req) + latency := time.Since(startTime) if respErr != nil { if errors.Is(respErr, context.Canceled) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: respErr, }, - } + }, latency) } // Check for timeout first using net.Error before checking net.OpError var netErr net.Error if errors.As(respErr, &netErr) && netErr.Timeout() { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, respErr) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, respErr), latency) } if errors.Is(respErr, http.ErrHandlerTimeout) || errors.Is(respErr, context.DeadlineExceeded) { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, respErr) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, respErr), latency) } // Check for DNS lookup and network errors after timeout checks var opErr *net.OpError var dnsErr *net.DNSError if errors.As(respErr, &opErr) || errors.As(respErr, &dnsErr) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Message: schemas.ErrProviderNetworkError, Error: respErr, }, - } + }, latency) } - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Message: schemas.ErrProviderDoRequest, Error: respErr, }, - } + }, latency) } // Extract provider response headers before status check so error responses also forward them @@ -575,51 +558,31 @@ func (provider *BedrockProvider) makeStreamingRequest(ctx *schemas.BifrostContex if resp.StatusCode != http.StatusOK { body, _ := io.ReadAll(resp.Body) resp.Body.Close() - return nil, parseBedrockHTTPError(resp.StatusCode, resp.Header, body) + return nil, providerUtils.SetErrorLatency(parseBedrockHTTPError(resp.StatusCode, resp.Header, body), latency) } return resp, nil } -// signAWSRequest signs an HTTP request using AWS Signature Version 4. -// It is used in providers like Bedrock. -// It sets required headers, calculates the request body hash, and signs the request -// using the provided AWS credentials. -// signAWSRequestFromKey is a convenience wrapper around signAWSRequest that reads -// credentials from a BedrockKeyConfig. When cfg is nil (no explicit key configured), -// all credential fields are zero-valued, causing signAWSRequest to fall back to the -// default AWS credential chain (IAM role, env vars, instance profile, etc.). -func signAWSRequestFromKey( - ctx *schemas.BifrostContext, - req *http.Request, - cfg *schemas.BedrockKeyConfig, - region, service string, -) *schemas.BifrostError { - if cfg != nil { - return signAWSRequest(ctx, req, - cfg.AccessKey, cfg.SecretKey, - cfg.SessionToken, cfg.RoleARN, - cfg.ExternalID, cfg.RoleSessionName, - region, service) - } - // No config: pass zero SecretVar values so signAWSRequest uses the default chain. - return signAWSRequest(ctx, req, - schemas.SecretVar{}, schemas.SecretVar{}, - nil, nil, nil, nil, - region, service) -} - // Returns a BifrostError if signing fails. func signAWSRequest( ctx *schemas.BifrostContext, req *http.Request, - accessKey, secretKey schemas.SecretVar, - sessionToken *schemas.SecretVar, - roleARN *schemas.SecretVar, - externalID *schemas.SecretVar, - sessionName *schemas.SecretVar, + keyCfg *schemas.BedrockKeyConfig, region, service string, ) *schemas.BifrostError { + var accessKey, secretKey schemas.SecretVar + var sessionToken, roleARN, externalID, sessionName *schemas.SecretVar + + if keyCfg != nil { + accessKey = keyCfg.AccessKey + secretKey = keyCfg.SecretKey + sessionToken = keyCfg.SessionToken + roleARN = keyCfg.RoleARN + externalID = keyCfg.ExternalID + sessionName = keyCfg.RoleSessionName + } + // Set required headers before signing (only if not already set) if req.Header.Get("Content-Type") == "" { req.Header.Set("Content-Type", "application/json") @@ -754,7 +717,7 @@ func signAWSRequest( // for this GET). Best-effort: returns nil on any failure so the foundation-model list is // still returned. func (provider *BedrockProvider) listMantleModels(ctx *schemas.BifrostContext, key schemas.Key, region string, unfiltered bool) *schemas.BifrostListModelsResponse { - mURL := mantleURL(region, "", "models") + mURL := mantleOpenAIURL(region, "", "models") req, err := http.NewRequestWithContext(ctx, http.MethodGet, mURL, nil) if err != nil { provider.logger.Warn("failed to build mantle list-models request: %v", err) @@ -763,7 +726,7 @@ func (provider *BedrockProvider) listMantleModels(ctx *schemas.BifrostContext, k providerUtils.SetExtraHeadersHTTP(ctx, req, provider.networkConfig.ExtraHeaders, nil) if key.Value.GetValue() != "" { req.Header.Set("Authorization", fmt.Sprintf("Bearer %s", key.Value.GetValue())) - } else if bifrostErr := signAWSRequestFromKey(ctx, req, key.BedrockKeyConfig, region, bedrockMantleSigningService); bifrostErr != nil { + } else if bifrostErr := signAWSRequest(ctx, req, key.BedrockKeyConfig, region, bedrockMantleSigningService); bifrostErr != nil { provider.logger.Warn("failed to sign mantle list-models request: %v", bifrostErr.Error.Message) return nil } @@ -841,7 +804,7 @@ func (provider *BedrockProvider) listModelsByKey(ctx *schemas.BifrostContext, ke } else { // Sign the request using either explicit credentials or IAM role authentication - if err := signAWSRequest(ctx, req, config.AccessKey, config.SecretKey, config.SessionToken, config.RoleARN, config.ExternalID, config.RoleSessionName, region, bedrockSigningService); err != nil { + if err := signAWSRequest(ctx, req, config, region, bedrockSigningService); err != nil { return nil, err } } @@ -850,61 +813,62 @@ func (provider *BedrockProvider) listModelsByKey(ctx *schemas.BifrostContext, ke // Execute the request resp, err := provider.client.Do(req) + latency := time.Since(startTime) if err != nil { if errors.Is(err, context.Canceled) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } // Check for timeout first using net.Error before checking net.OpError var netErr net.Error if errors.As(err, &netErr) && netErr.Timeout() { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } if errors.Is(err, http.ErrHandlerTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } // Check for DNS lookup and network errors after timeout checks var opErr *net.OpError var dnsErr *net.DNSError if errors.As(err, &opErr) || errors.As(err, &dnsErr) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Message: schemas.ErrProviderNetworkError, Error: err, }, - } + }, latency) } - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Message: schemas.ErrProviderDoRequest, Error: err, }, - } + }, latency) } // Read response body and close responseBody, err := io.ReadAll(resp.Body) resp.Body.Close() if err != nil { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: true, Error: &schemas.ErrorField{ Message: "error reading request", Error: err, }, - } + }, latency) } if resp.StatusCode != http.StatusOK { - return nil, parseBedrockHTTPError(resp.StatusCode, resp.Header, responseBody) + return nil, providerUtils.SetErrorLatency(parseBedrockHTTPError(resp.StatusCode, resp.Header, responseBody), latency) } // Parse Bedrock-specific response @@ -990,7 +954,7 @@ func (provider *BedrockProvider) TextCompletion(ctx *schemas.BifrostContext, key ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Handle model-specific response conversion @@ -1171,18 +1135,19 @@ func (provider *BedrockProvider) TextCompletionStream(ctx *schemas.BifrostContex } // ChatCompletion performs a chat completion request to Bedrock's API. -// It formats the request, sends it to Bedrock, and processes the response. +// OpenAI-family and Gemma 4 models route via the Bedrock Mantle OpenAI-compatible endpoint. +// All other models (including Anthropic/Claude) use the Bedrock Converse API. // Returns a BifrostResponse containing the completion results or an error if the request fails. func (provider *BedrockProvider) ChatCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostChatRequest) (*schemas.BifrostChatResponse, *schemas.BifrostError) { if err := providerUtils.CheckOperationAllowed(schemas.Bedrock, provider.customProviderConfig, schemas.ChatCompletionRequest); err != nil { return nil, err } - if isMantleModel(schemas.ResolveCanonicalModel(ctx, request.Model)) { - return provider.chatCompletionViaMantle(ctx, key, request) + if isMantleModel(ctx, request.Model) { + return provider.mantleChatCompletions(ctx, key, request) } - // Use centralized Bedrock converter + // Use Bedrock Converse API for all other models jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody( ctx, request, @@ -1192,8 +1157,6 @@ func (provider *BedrockProvider) ChatCompletion(ctx *schemas.BifrostContext, key if bifrostErr != nil { return nil, bifrostErr } - - // Format the path with proper model identifier path, _ := provider.getModelPathAndRegion(ctx, "converse", request.Model, key) // Create the signed request @@ -1202,26 +1165,25 @@ func (provider *BedrockProvider) ChatCompletion(ctx *schemas.BifrostContext, key ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } - // pool the response + // Parse Bedrock Converse API response bedrockResponse := acquireBedrockChatResponse() defer releaseBedrockChatResponse(bedrockResponse) // Parse the response using the new Bedrock type if err := sonic.Unmarshal(responseBody, bedrockResponse); err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("failed to parse bedrock response", err), jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("failed to parse bedrock response", err), jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Convert using the new response converter bifrostResponse, err := bedrockResponse.ToBifrostChatResponse(ctx, request.Model) if err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("failed to convert bedrock response", err), jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("failed to convert bedrock response", err), jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } - // Override finish reason for structured output - // When structured output is used, tool_use is expected but should appear as "stop" to the client + // Override finish reason for structured output (Converse API only) if _, ok := ctx.Value(schemas.BifrostContextKeyStructuredOutputToolName).(string); ok { if len(bifrostResponse.Choices) > 0 && bifrostResponse.Choices[0].FinishReason != nil { if *bifrostResponse.Choices[0].FinishReason == string(schemas.BifrostFinishReasonToolCalls) { @@ -1234,12 +1196,9 @@ func (provider *BedrockProvider) ChatCompletion(ctx *schemas.BifrostContext, key bifrostResponse.ExtraFields.Latency = latency.Milliseconds() bifrostResponse.ExtraFields.ProviderResponseHeaders = providerResponseHeaders - // Set raw request if enabled if providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest) { providerUtils.ParseAndSetRawRequest(&bifrostResponse.ExtraFields, jsonData) } - - // Set raw response if enabled if providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) { var rawResponse interface{} if err := sonic.Unmarshal(responseBody, &rawResponse); err == nil { @@ -1262,18 +1221,94 @@ func normalizeCachedUsage(usage *schemas.BifrostLLMUsage) { usage.PromptTokens += usage.PromptTokensDetails.CachedReadTokens + usage.PromptTokensDetails.CachedWriteTokens } +func accumulateBedrockResponsesUsage(usage *schemas.ResponsesResponseUsage, billedUsage *schemas.BifrostLLMUsage, usageToProcess *BedrockTokenUsage) { + if usage == nil || usageToProcess == nil { + return + } + if usageToProcess.InputTokens > usage.InputTokens { + usage.InputTokens = usageToProcess.InputTokens + if billedUsage != nil { + billedUsage.PromptTokens = usageToProcess.InputTokens + } + } + if usageToProcess.OutputTokens > usage.OutputTokens { + usage.OutputTokens = usageToProcess.OutputTokens + if billedUsage != nil { + billedUsage.CompletionTokens = usageToProcess.OutputTokens + } + } + if usageToProcess.TotalTokens > usage.TotalTokens { + usage.TotalTokens = usageToProcess.TotalTokens + if billedUsage != nil { + billedUsage.TotalTokens = usageToProcess.TotalTokens + } + } + if usageToProcess.CacheReadInputTokens > 0 { + if usage.InputTokensDetails == nil { + usage.InputTokensDetails = &schemas.ResponsesResponseInputTokens{} + } + if billedUsage != nil && billedUsage.PromptTokensDetails == nil { + billedUsage.PromptTokensDetails = &schemas.ChatPromptTokensDetails{} + } + if usageToProcess.CacheReadInputTokens > usage.InputTokensDetails.CachedReadTokens { + usage.InputTokensDetails.CachedReadTokens = usageToProcess.CacheReadInputTokens + if billedUsage != nil { + billedUsage.PromptTokensDetails.CachedReadTokens = usageToProcess.CacheReadInputTokens + } + } + } + if usageToProcess.CacheWriteInputTokens > 0 { + if usage.InputTokensDetails == nil { + usage.InputTokensDetails = &schemas.ResponsesResponseInputTokens{} + } + if billedUsage != nil && billedUsage.PromptTokensDetails == nil { + billedUsage.PromptTokensDetails = &schemas.ChatPromptTokensDetails{} + } + if usageToProcess.CacheWriteInputTokens > usage.InputTokensDetails.CachedWriteTokens { + usage.InputTokensDetails.CachedWriteTokens = usageToProcess.CacheWriteInputTokens + if billedUsage != nil { + billedUsage.PromptTokensDetails.CachedWriteTokens = usageToProcess.CacheWriteInputTokens + } + } + if usageToProcess.CacheDetails != nil { + if usage.InputTokensDetails.CachedWriteTokenDetails == nil { + usage.InputTokensDetails.CachedWriteTokenDetails = &schemas.ChatCachedWriteTokenDetails{} + } + if billedUsage != nil && billedUsage.PromptTokensDetails.CachedWriteTokenDetails == nil { + billedUsage.PromptTokensDetails.CachedWriteTokenDetails = &schemas.ChatCachedWriteTokenDetails{} + } + for _, cacheDetail := range *usageToProcess.CacheDetails { + if cacheDetail.TTL == BedrockCacheWriteTTL5m { + usage.InputTokensDetails.CachedWriteTokenDetails.CachedWriteTokens5m = cacheDetail.InputTokens + if billedUsage != nil { + billedUsage.PromptTokensDetails.CachedWriteTokenDetails.CachedWriteTokens5m = cacheDetail.InputTokens + } + } + if cacheDetail.TTL == BedrockCacheWriteTTL1h { + usage.InputTokensDetails.CachedWriteTokenDetails.CachedWriteTokens1h = cacheDetail.InputTokens + if billedUsage != nil { + billedUsage.PromptTokensDetails.CachedWriteTokenDetails.CachedWriteTokens1h = cacheDetail.InputTokens + } + } + } + } + } +} + // ChatCompletionStream performs a streaming chat completion request to Bedrock's API. -// It formats the request, sends it to Bedrock, and processes the streaming response. +// OpenAI-family and Gemma 4 models route via the Bedrock Mantle OpenAI-compatible endpoint. +// All other models (including Anthropic/Claude) use the Bedrock Converse streaming API. // Returns a channel for streaming BifrostStreamChunk objects or an error if the request fails. func (provider *BedrockProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { if err := providerUtils.CheckOperationAllowed(schemas.Bedrock, provider.customProviderConfig, schemas.ChatCompletionStreamRequest); err != nil { return nil, err } - if isMantleModel(schemas.ResolveCanonicalModel(ctx, request.Model)) { - return provider.chatCompletionStreamViaMantle(ctx, postHookRunner, postHookSpanFinalizer, key, request) + if isMantleModel(ctx, request.Model) { + return provider.mantleChatCompletionsStream(ctx, postHookRunner, postHookSpanFinalizer, key, request) } + // Use Bedrock Converse streaming API for all other models jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody( ctx, request, @@ -1285,6 +1320,7 @@ func (provider *BedrockProvider) ChatCompletionStream(ctx *schemas.BifrostContex } startTime := time.Now() + resp, bifrostErr := provider.makeStreamingRequest(ctx, jsonData, key, request.Model, "converse-stream") if bifrostErr != nil { return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) @@ -1358,13 +1394,12 @@ func (provider *BedrockProvider) ChatCompletionStream(ctx *schemas.BifrostContex if toolName, ok := ctx.Value(schemas.BifrostContextKeyStructuredOutputToolName).(string); ok { structuredOutputToolName = toolName } - var structuredOutputBuilder strings.Builder - var isAccumulatingStructuredOutput bool streamState := NewBedrockStreamStateWithContext(ctx) + var isAccumulatingStructuredOutput bool + var structuredOutputBuilder strings.Builder for { - // If context was cancelled/timed out, let defer handle it if ctx.Err() != nil { return } @@ -1413,7 +1448,7 @@ func (provider *BedrockProvider) ChatCompletionStream(ctx *schemas.BifrostContex } } - // Parse the JSON event into our typed structure + // Converse API path: parse Bedrock Converse-specific stream events var streamEvent BedrockStreamEvent if err := sonic.Unmarshal(message.Payload, &streamEvent); err != nil { provider.logger.Debug("Failed to parse JSON from event buffer: %v, data: %s", err, string(message.Payload)) @@ -1561,7 +1596,7 @@ func (provider *BedrockProvider) ChatCompletionStream(ctx *schemas.BifrostContex normalizeUsage() - // Send final response + // Send final chunk with accumulated usage response := providerUtils.CreateBifrostChatCompletionChunkResponse(id, usage, finishReason, chunkIndex, request.Model, 0) // Set raw request if enabled if providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest) { @@ -1575,19 +1610,20 @@ func (provider *BedrockProvider) ChatCompletionStream(ctx *schemas.BifrostContex return responseChan, nil } -// Responses performs a chat completion request to Anthropic's API. -// It formats the request, sends it to Anthropic, and processes the response. +// Responses performs a responses request to Bedrock's API. +// OpenAI-family and Gemma 4 models route via the Bedrock Mantle OpenAI-compatible endpoint. +// All other models (including Anthropic/Claude) use the Bedrock Converse API. // Returns a BifrostResponse containing the completion results or an error if the request fails. func (provider *BedrockProvider) Responses(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostResponsesRequest) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) { if err := providerUtils.CheckOperationAllowed(schemas.Bedrock, provider.customProviderConfig, schemas.ResponsesRequest); err != nil { return nil, err } - if isMantleModel(schemas.ResolveCanonicalModel(ctx, request.Model)) { - return provider.responsesViaMantle(ctx, key, request) + if isMantleModel(ctx, request.Model) { + return provider.mantleResponses(ctx, key, request) } - // Use centralized Bedrock converter + // Use Bedrock Converse API for all other models jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody( ctx, request, @@ -1597,8 +1633,6 @@ func (provider *BedrockProvider) Responses(ctx *schemas.BifrostContext, key sche if bifrostErr != nil { return nil, bifrostErr } - - // Format the path with proper model identifier path, _ := provider.getModelPathAndRegion(ctx, "converse", request.Model, key) // Create the signed request @@ -1607,22 +1641,22 @@ func (provider *BedrockProvider) Responses(ctx *schemas.BifrostContext, key sche ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } - // pool the response + // Parse Bedrock Converse API response bedrockResponse := acquireBedrockChatResponse() defer releaseBedrockChatResponse(bedrockResponse) // Parse the response using the new Bedrock type if err := sonic.Unmarshal(responseBody, bedrockResponse); err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("failed to parse bedrock response", err), jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("failed to parse bedrock response", err), jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Convert using the new response converter bifrostResponse, err := bedrockResponse.ToBifrostResponsesResponse(ctx) if err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("failed to convert bedrock response", err), jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("failed to convert bedrock response", err), jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse.Model = request.Model @@ -1655,10 +1689,11 @@ func (provider *BedrockProvider) ResponsesStream(ctx *schemas.BifrostContext, po return nil, err } - if isMantleModel(schemas.ResolveCanonicalModel(ctx, request.Model)) { - return provider.responsesStreamViaMantle(ctx, postHookRunner, postHookSpanFinalizer, key, request) + if isMantleModel(ctx, request.Model) { + return provider.mantleResponsesStream(ctx, postHookRunner, postHookSpanFinalizer, key, request) } + // Use Bedrock Converse streaming API for all other models jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody( ctx, request, @@ -1670,9 +1705,11 @@ func (provider *BedrockProvider) ResponsesStream(ctx *schemas.BifrostContext, po } startTime := time.Now() + resp, bifrostErr := provider.makeStreamingRequest(ctx, jsonData, key, request.Model, "converse-stream") + latency := time.Since(startTime) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerUtils.ExtractProviderResponseHeadersFromHTTP(resp)) @@ -1706,10 +1743,30 @@ func (provider *BedrockProvider) ResponsesStream(ctx *schemas.BifrostContext, po // Process AWS Event Stream format usage := &schemas.ResponsesResponseUsage{} + billedUsage := &schemas.BifrostLLMUsage{} + // Register the accumulating usage handle so a mid-stream cancel/timeout + // can bill for Bedrock Responses usage already reported by stream events + // before the stream was interrupted. + ctx.SetValue(schemas.BifrostContextKeyStreamAccumulatedUsage, billedUsage) + + usageNormalized := false + normalizeUsage := func() { + if usageNormalized { + return + } + usageNormalized = true + normalizeCachedUsage(billedUsage) + } + defer func() { + if ctx.Err() != nil { + normalizeUsage() + } + }() + var streamTrace *BedrockConverseTrace chunkIndex := 0 - // Create stream state for stateful conversions + // Create stream state for stateful conversions (used by Converse API path) streamState := acquireBedrockResponsesStreamState() streamState.Model = &request.Model streamState.Ctx = ctx @@ -1740,7 +1797,7 @@ func (provider *BedrockProvider) ResponsesStream(ctx *schemas.BifrostContext, po return } if err == io.EOF { - // End of stream - finalize any open items + // Converse API: finalize any open items at end of stream. finalResponses := FinalizeBedrockStream(streamState, chunkIndex, usage, streamTrace) for i, finalResponse := range finalResponses { finalResponse.ExtraFields = schemas.BifrostResponseExtraFields{ @@ -1801,7 +1858,7 @@ func (provider *BedrockProvider) ResponsesStream(ctx *schemas.BifrostContext, po } } - // Parse the JSON event into our typed structure + // Converse API path: parse Bedrock Converse-specific stream events var streamEvent BedrockStreamEvent if err := sonic.Unmarshal(message.Payload, &streamEvent); err != nil { provider.logger.Debug("Failed to parse JSON from event buffer: %v, data: %s", err, string(message.Payload)) @@ -1816,45 +1873,7 @@ func (provider *BedrockProvider) ResponsesStream(ctx *schemas.BifrostContext, po if streamEvent.Usage != nil { // Accumulate usage information instead of overwriting // In some cases usage comes in multiple events, so we need to take the maximum values - if streamEvent.Usage.InputTokens > usage.InputTokens { - usage.InputTokens = streamEvent.Usage.InputTokens - } - if streamEvent.Usage.OutputTokens > usage.OutputTokens { - usage.OutputTokens = streamEvent.Usage.OutputTokens - } - if streamEvent.Usage.TotalTokens > usage.TotalTokens { - usage.TotalTokens = streamEvent.Usage.TotalTokens - } - // Handle cached tokens if present - if streamEvent.Usage.CacheReadInputTokens > 0 { - if usage.InputTokensDetails == nil { - usage.InputTokensDetails = &schemas.ResponsesResponseInputTokens{} - } - if streamEvent.Usage.CacheReadInputTokens > usage.InputTokensDetails.CachedReadTokens { - usage.InputTokensDetails.CachedReadTokens = streamEvent.Usage.CacheReadInputTokens - } - } - if streamEvent.Usage.CacheWriteInputTokens > 0 { - if usage.InputTokensDetails == nil { - usage.InputTokensDetails = &schemas.ResponsesResponseInputTokens{} - } - if streamEvent.Usage.CacheWriteInputTokens > usage.InputTokensDetails.CachedWriteTokens { - usage.InputTokensDetails.CachedWriteTokens = streamEvent.Usage.CacheWriteInputTokens - } - if streamEvent.Usage.CacheDetails != nil { - if usage.InputTokensDetails.CachedWriteTokenDetails == nil { - usage.InputTokensDetails.CachedWriteTokenDetails = &schemas.ChatCachedWriteTokenDetails{} - } - for _, cacheDetail := range *streamEvent.Usage.CacheDetails { - if cacheDetail.TTL == BedrockCacheWriteTTL5m { - usage.InputTokensDetails.CachedWriteTokenDetails.CachedWriteTokens5m = cacheDetail.InputTokens - } - if cacheDetail.TTL == BedrockCacheWriteTTL1h { - usage.InputTokensDetails.CachedWriteTokenDetails.CachedWriteTokens1h = cacheDetail.InputTokens - } - } - } - } + accumulateBedrockResponsesUsage(usage, billedUsage, streamEvent.Usage) } // Handle structured output: intercept tool calls for the structured output tool @@ -1992,7 +2011,7 @@ func (provider *BedrockProvider) Embedding(ctx *schemas.BifrostContext, key sche ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if bifrostError != nil { - return nil, providerUtils.EnrichError(ctx, bifrostError, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostError, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Parse response based on model type var bifrostResponse *schemas.BifrostEmbeddingResponse @@ -2000,7 +2019,7 @@ func (provider *BedrockProvider) Embedding(ctx *schemas.BifrostContext, key sche case "titan": var titanResp BedrockTitanEmbeddingResponse if err := sonic.Unmarshal(rawResponse, &titanResp); err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("error parsing Titan embedding response", err), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("error parsing Titan embedding response", err), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse = titanResp.ToBifrostEmbeddingResponse() bifrostResponse.Model = request.Model @@ -2008,11 +2027,11 @@ func (provider *BedrockProvider) Embedding(ctx *schemas.BifrostContext, key sche case "cohere": var cohereResp BedrockCohereEmbeddingResponse if err := sonic.Unmarshal(rawResponse, &cohereResp); err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("error parsing Cohere embedding response", err), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("error parsing Cohere embedding response", err), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } converted, convErr := cohereResp.ToBifrostEmbeddingResponse() if convErr != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("error parsing Cohere embedding response", convErr), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("error parsing Cohere embedding response", convErr), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse = converted bifrostResponse.Model = request.Model @@ -2085,13 +2104,13 @@ func (provider *BedrockProvider) Rerank(ctx *schemas.BifrostContext, key schemas ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } response := &BedrockRerankResponse{} rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(rawResponseBody, response, jsonData, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, rawResponseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, rawResponseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } returnDocuments := request.Params != nil && request.Params.ReturnDocuments != nil && *request.Params.ReturnDocuments @@ -2176,18 +2195,18 @@ func (provider *BedrockProvider) ImageGeneration(ctx *schemas.BifrostContext, ke ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if bifrostError != nil { - return nil, providerUtils.EnrichError(ctx, bifrostError, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostError, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Parse response based on model type var bifrostResponse *schemas.BifrostImageGenerationResponse var imageResp BedrockImageGenerationResponse if err := sonic.Unmarshal(rawResponse, &imageResp); err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("error parsing image generation response", err), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("error parsing image generation response", err), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if imageResp.Error != "" { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(imageResp.Error, nil), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(imageResp.Error, nil), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse = ToBifrostImageGenerationResponse(&imageResp) @@ -2251,17 +2270,17 @@ func (provider *BedrockProvider) ImageEdit(ctx *schemas.BifrostContext, key sche ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if bifrostError != nil { - return nil, providerUtils.EnrichError(ctx, bifrostError, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostError, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Parse response (reuse BedrockImageGenerationResponse) var imageResp BedrockImageGenerationResponse if err := sonic.Unmarshal(rawResponse, &imageResp); err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("error parsing image edit response", err), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("error parsing image edit response", err), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if imageResp.Error != "" { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(imageResp.Error, nil), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(imageResp.Error, nil), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Convert response and set metadata @@ -2318,17 +2337,17 @@ func (provider *BedrockProvider) ImageVariation(ctx *schemas.BifrostContext, key ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if bifrostError != nil { - return nil, providerUtils.EnrichError(ctx, bifrostError, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostError, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Parse response (reuse BedrockImageGenerationResponse and ToBifrostImageGenerationResponse) var imageResp BedrockImageGenerationResponse if err := sonic.Unmarshal(rawResponse, &imageResp); err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("error parsing image variation response", err), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("error parsing image variation response", err), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if imageResp.Error != "" { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(imageResp.Error, nil), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(imageResp.Error, nil), jsonData, rawResponse, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Convert response and set metadata @@ -2454,7 +2473,7 @@ func (provider *BedrockProvider) FileUpload(ctx *schemas.BifrostContext, key sch httpReq.ContentLength = int64(len(request.File)) // Sign request for S3 - if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig.AccessKey, key.BedrockKeyConfig.SecretKey, key.BedrockKeyConfig.SessionToken, key.BedrockKeyConfig.RoleARN, key.BedrockKeyConfig.ExternalID, key.BedrockKeyConfig.RoleSessionName, region, "s3"); err != nil { + if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig, region, "s3"); err != nil { provider.logger.Error("error signing request: %s", err.Error.Message) return nil, err } @@ -2584,7 +2603,7 @@ func (provider *BedrockProvider) FileList(ctx *schemas.BifrostContext, keys []sc } // Sign request for S3 - if bifrostErr := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig.AccessKey, key.BedrockKeyConfig.SecretKey, key.BedrockKeyConfig.SessionToken, key.BedrockKeyConfig.RoleARN, key.BedrockKeyConfig.ExternalID, key.BedrockKeyConfig.RoleSessionName, region, "s3"); bifrostErr != nil { + if bifrostErr := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig, region, "s3"); bifrostErr != nil { return nil, bifrostErr } @@ -2695,7 +2714,7 @@ func (provider *BedrockProvider) FileRetrieve(ctx *schemas.BifrostContext, keys } // Sign request for S3 - if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig.AccessKey, key.BedrockKeyConfig.SecretKey, key.BedrockKeyConfig.SessionToken, key.BedrockKeyConfig.RoleARN, key.BedrockKeyConfig.ExternalID, key.BedrockKeyConfig.RoleSessionName, region, "s3"); err != nil { + if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig, region, "s3"); err != nil { lastErr = err continue } @@ -2793,7 +2812,7 @@ func (provider *BedrockProvider) FileDelete(ctx *schemas.BifrostContext, keys [] } // Sign request for S3 - if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig.AccessKey, key.BedrockKeyConfig.SecretKey, key.BedrockKeyConfig.SessionToken, key.BedrockKeyConfig.RoleARN, key.BedrockKeyConfig.ExternalID, key.BedrockKeyConfig.RoleSessionName, region, "s3"); err != nil { + if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig, region, "s3"); err != nil { lastErr = err continue } @@ -2874,7 +2893,7 @@ func (provider *BedrockProvider) FileContent(ctx *schemas.BifrostContext, keys [ } // Sign request for S3 - if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig.AccessKey, key.BedrockKeyConfig.SecretKey, key.BedrockKeyConfig.SessionToken, key.BedrockKeyConfig.RoleARN, key.BedrockKeyConfig.ExternalID, key.BedrockKeyConfig.RoleSessionName, region, "s3"); err != nil { + if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig, region, "s3"); err != nil { lastErr = err continue } @@ -3080,7 +3099,7 @@ func (provider *BedrockProvider) BatchCreate(ctx *schemas.BifrostContext, key sc } // Sign request - if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig.AccessKey, key.BedrockKeyConfig.SecretKey, key.BedrockKeyConfig.SessionToken, key.BedrockKeyConfig.RoleARN, key.BedrockKeyConfig.ExternalID, key.BedrockKeyConfig.RoleSessionName, region, bedrockSigningService); err != nil { + if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig, region, bedrockSigningService); err != nil { return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, sendBackRawRequest, sendBackRawResponse) } @@ -3205,7 +3224,7 @@ func (provider *BedrockProvider) BatchList(ctx *schemas.BifrostContext, keys []s } // Sign request - if bifrostErr := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig.AccessKey, key.BedrockKeyConfig.SecretKey, key.BedrockKeyConfig.SessionToken, key.BedrockKeyConfig.RoleARN, key.BedrockKeyConfig.ExternalID, key.BedrockKeyConfig.RoleSessionName, region, bedrockSigningService); bifrostErr != nil { + if bifrostErr := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig, region, bedrockSigningService); bifrostErr != nil { return nil, bifrostErr } @@ -3324,7 +3343,7 @@ func (provider *BedrockProvider) fetchBatchManifest(ctx *schemas.BifrostContext, } // Sign request for S3 - if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig.AccessKey, key.BedrockKeyConfig.SecretKey, key.BedrockKeyConfig.SessionToken, key.BedrockKeyConfig.RoleARN, key.BedrockKeyConfig.ExternalID, key.BedrockKeyConfig.RoleSessionName, region, "s3"); err != nil { + if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig, region, "s3"); err != nil { provider.logger.Error("failed to sign manifest request: %v", err) return nil } @@ -3384,7 +3403,7 @@ func (provider *BedrockProvider) BatchRetrieve(ctx *schemas.BifrostContext, keys } // Sign request - if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig.AccessKey, key.BedrockKeyConfig.SecretKey, key.BedrockKeyConfig.SessionToken, key.BedrockKeyConfig.RoleARN, key.BedrockKeyConfig.ExternalID, key.BedrockKeyConfig.RoleSessionName, region, bedrockSigningService); err != nil { + if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig, region, bedrockSigningService); err != nil { lastErr = err continue } @@ -3528,7 +3547,7 @@ func (provider *BedrockProvider) BatchCancel(ctx *schemas.BifrostContext, keys [ } // Sign request - if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig.AccessKey, key.BedrockKeyConfig.SecretKey, key.BedrockKeyConfig.SessionToken, key.BedrockKeyConfig.RoleARN, key.BedrockKeyConfig.ExternalID, key.BedrockKeyConfig.RoleSessionName, region, bedrockSigningService); err != nil { + if err := signAWSRequest(ctx, httpReq, key.BedrockKeyConfig, region, bedrockSigningService); err != nil { lastErr = err continue } @@ -3784,7 +3803,7 @@ func (provider *BedrockProvider) CountTokens(ctx *schemas.BifrostContext, key sc }, }, nil } - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Parse the response @@ -3797,7 +3816,7 @@ func (provider *BedrockProvider) CountTokens(ctx *schemas.BifrostContext, key sc providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), ) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Convert to Bifrost format diff --git a/core/providers/bedrock/bedrock_test.go b/core/providers/bedrock/bedrock_test.go index 4d67a334fea..5eba65c9c32 100644 --- a/core/providers/bedrock/bedrock_test.go +++ b/core/providers/bedrock/bedrock_test.go @@ -4594,17 +4594,19 @@ func TestBedrockStopReasonMappingResponsesPath(t *testing.T) { t.Parallel() tests := []struct { - name string - bedrockReason string - expectedBifrost string + name string + bedrockReason string + expectedBifrost string + expectedStatus string // "" means Status should be nil + expectedIncompleteDetails string // "" means IncompleteDetails should be nil }{ - {"EndTurn", "end_turn", "stop"}, - {"MaxTokens", "max_tokens", "length"}, - {"StopSequence", "stop_sequence", "stop"}, - {"ToolUse", "tool_use", "tool_calls"}, - {"ContentFiltered", "content_filtered", "content_filter"}, - {"GuardrailIntervened", "guardrail_intervened", "guardrail_intervened"}, // no clean mapping — passes through - {"UnknownReason", "some_unknown_reason", "some_unknown_reason"}, // no clean mapping — passes through + {"EndTurn", "end_turn", "stop", "completed", ""}, + {"MaxTokens", "max_tokens", "length", "incomplete", "max_output_tokens"}, + {"StopSequence", "stop_sequence", "stop", "completed", ""}, + {"ToolUse", "tool_use", "tool_calls", "completed", ""}, + {"ContentFiltered", "content_filtered", "content_filter", "", ""}, // no clean mapping — passes through, no Status + {"GuardrailIntervened", "guardrail_intervened", "guardrail_intervened", "", ""}, // no clean mapping — passes through, no Status + {"UnknownReason", "some_unknown_reason", "some_unknown_reason", "", ""}, // no clean mapping — passes through, no Status } ctx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline) @@ -4629,10 +4631,88 @@ func TestBedrockStopReasonMappingResponsesPath(t *testing.T) { require.NotNil(t, bifrostResp.StopReason, "StopReason should be set") assert.Equal(t, tt.expectedBifrost, *bifrostResp.StopReason, "Bedrock stop reason %q should map to %q in responses path", tt.bedrockReason, tt.expectedBifrost) + + // Status + IncompleteDetails mirror the OpenAI Responses-API spec. + // Truncation must surface as Status="incomplete" with the canonical + // IncompleteDetails.Reason; clean completions get Status="completed"; + // unmapped reasons leave both unset (preserves prior behavior). + if tt.expectedStatus == "" { + assert.Nil(t, bifrostResp.Status, "Status must be nil for unmapped stop reasons") + assert.Nil(t, bifrostResp.IncompleteDetails, "IncompleteDetails must be nil for unmapped stop reasons") + } else { + require.NotNil(t, bifrostResp.Status, "Status must be set for mapped stop reason %q", tt.bedrockReason) + assert.Equal(t, tt.expectedStatus, *bifrostResp.Status) + } + if tt.expectedIncompleteDetails == "" { + assert.Nil(t, bifrostResp.IncompleteDetails) + } else { + require.NotNil(t, bifrostResp.IncompleteDetails) + assert.Equal(t, tt.expectedIncompleteDetails, bifrostResp.IncompleteDetails.Reason) + } }) } } +// TestFinalizeBedrockStream_MaxTokensTruncation guards the streaming counterpart: +// when Bedrock's stopReason is "max_tokens", the terminal SSE event must be +// response.incomplete (not response.completed) and the embedded response must +// carry Status="incomplete" + IncompleteDetails.Reason="max_output_tokens" so +// streaming consumers can detect truncation. +func TestFinalizeBedrockStream_MaxTokensTruncation(t *testing.T) { + state := bedrock.NewBedrockResponsesStreamState() + state.StopReason = schemas.Ptr("length") // mapped from bedrock's "max_tokens" + usage := &schemas.ResponsesResponseUsage{InputTokens: 30, OutputTokens: 15, TotalTokens: 45} + + finalResponses := bedrock.FinalizeBedrockStream(state, 0, usage, nil) + require.NotEmpty(t, finalResponses) + + terminal := finalResponses[len(finalResponses)-1] + assert.Equal(t, schemas.ResponsesStreamResponseTypeIncomplete, terminal.Type, + "terminal event must be response.incomplete on max_output_tokens truncation") + require.NotNil(t, terminal.Response) + require.NotNil(t, terminal.Response.Status) + assert.Equal(t, "incomplete", *terminal.Response.Status) + require.NotNil(t, terminal.Response.IncompleteDetails) + assert.Equal(t, "max_output_tokens", terminal.Response.IncompleteDetails.Reason) +} + +// TestFinalizeBedrockStream_CleanCompletionUnaffected verifies non-truncation +// stop reasons still produce response.completed with Status="completed". +func TestFinalizeBedrockStream_CleanCompletionUnaffected(t *testing.T) { + state := bedrock.NewBedrockResponsesStreamState() + state.StopReason = schemas.Ptr("stop") + usage := &schemas.ResponsesResponseUsage{InputTokens: 5, OutputTokens: 10, TotalTokens: 15} + + finalResponses := bedrock.FinalizeBedrockStream(state, 0, usage, nil) + require.NotEmpty(t, finalResponses) + + terminal := finalResponses[len(finalResponses)-1] + assert.Equal(t, schemas.ResponsesStreamResponseTypeCompleted, terminal.Type) + require.NotNil(t, terminal.Response) + require.NotNil(t, terminal.Response.Status) + assert.Equal(t, "completed", *terminal.Response.Status) + assert.Nil(t, terminal.Response.IncompleteDetails) +} + +// TestFinalizeBedrockStream_UnmappedReasonLeavesStatusUnset keeps the streaming +// path aligned with the non-streaming mapping: an unmapped stop reason (e.g. +// content_filter) ends the stream as response.completed but must leave Status +// unset rather than asserting "completed". +func TestFinalizeBedrockStream_UnmappedReasonLeavesStatusUnset(t *testing.T) { + state := bedrock.NewBedrockResponsesStreamState() + state.StopReason = schemas.Ptr("content_filter") + usage := &schemas.ResponsesResponseUsage{InputTokens: 5, OutputTokens: 10, TotalTokens: 15} + + finalResponses := bedrock.FinalizeBedrockStream(state, 0, usage, nil) + require.NotEmpty(t, finalResponses) + + terminal := finalResponses[len(finalResponses)-1] + assert.Equal(t, schemas.ResponsesStreamResponseTypeCompleted, terminal.Type) + require.NotNil(t, terminal.Response) + assert.Nil(t, terminal.Response.Status, "unmapped stop reasons must leave Status unset, matching the non-streaming path") + assert.Nil(t, terminal.Response.IncompleteDetails) +} + // TestBifrostToBedrockStopReasonReverseMapping tests the reverse conversion // (BifrostResponsesResponse.StopReason → BedrockConverseResponse.StopReason). func TestBifrostToBedrockStopReasonReverseMapping(t *testing.T) { diff --git a/core/providers/bedrock/cancelbilling_test.go b/core/providers/bedrock/cancelbilling_test.go index 594f5f0be25..ce7a2885708 100644 --- a/core/providers/bedrock/cancelbilling_test.go +++ b/core/providers/bedrock/cancelbilling_test.go @@ -39,3 +39,63 @@ func TestNormalizeCachedUsage_NoCacheDetailsIsNoOp(t *testing.T) { func TestNormalizeCachedUsage_NilSafe(t *testing.T) { normalizeCachedUsage(nil) // must not panic } + +func TestAccumulateBedrockResponsesUsage_MirrorsCacheIntoBilledUsage(t *testing.T) { + responseUsage := &schemas.ResponsesResponseUsage{} + billedUsage := &schemas.BifrostLLMUsage{} + cacheDetails := []BedrockCacheWriteDetails{ + {TTL: BedrockCacheWriteTTL1h, InputTokens: 700}, + } + upstreamUsage := &BedrockTokenUsage{ + InputTokens: 10, + OutputTokens: 5, + TotalTokens: 1515, + CacheReadInputTokens: 800, + CacheWriteInputTokens: 700, + CacheDetails: &cacheDetails, + } + + accumulateBedrockResponsesUsage(responseUsage, billedUsage, upstreamUsage) + + if responseUsage.InputTokens != 10 || responseUsage.OutputTokens != 5 || responseUsage.TotalTokens != 1515 { + t.Fatalf("unexpected response usage: %+v", responseUsage) + } + if responseUsage.InputTokensDetails == nil { + t.Fatal("expected response cache details") + } + if responseUsage.InputTokensDetails.CachedReadTokens != 800 { + t.Fatalf("response cached read = %d, want 800", responseUsage.InputTokensDetails.CachedReadTokens) + } + if responseUsage.InputTokensDetails.CachedWriteTokens != 700 { + t.Fatalf("response cached write = %d, want 700", responseUsage.InputTokensDetails.CachedWriteTokens) + } + if responseUsage.InputTokensDetails.CachedWriteTokenDetails == nil || + responseUsage.InputTokensDetails.CachedWriteTokenDetails.CachedWriteTokens1h != 700 { + t.Fatalf("response cached write details = %+v, want 1h=700", responseUsage.InputTokensDetails.CachedWriteTokenDetails) + } + + if billedUsage.PromptTokens != 10 || billedUsage.CompletionTokens != 5 || billedUsage.TotalTokens != 1515 { + t.Fatalf("unexpected billed usage before normalization: %+v", billedUsage) + } + if billedUsage.PromptTokensDetails == nil { + t.Fatal("expected billed cache details") + } + if billedUsage.PromptTokensDetails.CachedReadTokens != 800 { + t.Fatalf("billed cached read = %d, want 800", billedUsage.PromptTokensDetails.CachedReadTokens) + } + if billedUsage.PromptTokensDetails.CachedWriteTokens != 700 { + t.Fatalf("billed cached write = %d, want 700", billedUsage.PromptTokensDetails.CachedWriteTokens) + } + if billedUsage.PromptTokensDetails.CachedWriteTokenDetails == nil || + billedUsage.PromptTokensDetails.CachedWriteTokenDetails.CachedWriteTokens1h != 700 { + t.Fatalf("billed cached write details = %+v, want 1h=700", billedUsage.PromptTokensDetails.CachedWriteTokenDetails) + } + + normalizeCachedUsage(billedUsage) + if billedUsage.PromptTokens != 1510 { + t.Fatalf("normalized billed prompt = %d, want 1510", billedUsage.PromptTokens) + } + if billedUsage.TotalTokens != 1515 { + t.Fatalf("normalized billed total = %d, want 1515", billedUsage.TotalTokens) + } +} diff --git a/core/providers/bedrock/conversestreamstop_test.go b/core/providers/bedrock/conversestreamstop_test.go new file mode 100644 index 00000000000..7452c2ce4b9 --- /dev/null +++ b/core/providers/bedrock/conversestreamstop_test.go @@ -0,0 +1,201 @@ +package bedrock + +import ( + "reflect" + "testing" + + "github.com/bytedance/sonic" + "github.com/maximhq/bifrost/core/schemas" +) + +// encodeConverseStream runs each Bifrost stream event through the Converse stream +// converter and flattens the results into wire-ready encoded events, in order, +// the same way the HTTP transport does before AWS Event Stream encoding. +func encodeConverseStream(t *testing.T, chunks []*schemas.BifrostResponsesStreamResponse) []BedrockEncodedEvent { + t.Helper() + var events []BedrockEncodedEvent + for i, chunk := range chunks { + bedrockEvent, err := ToBedrockConverseStreamResponse(chunk) + if err != nil { + t.Fatalf("chunk %d (%s): unexpected error: %v", i, chunk.Type, err) + } + if bedrockEvent == nil { + continue + } + events = append(events, bedrockEvent.ToEncodedEvents()...) + } + return events +} + +func encodedEventTypes(events []BedrockEncodedEvent) []string { + types := make([]string, 0, len(events)) + for _, event := range events { + types = append(types, event.EventType) + } + return types +} + +func contentBlockStopPayloads(t *testing.T, events []BedrockEncodedEvent) []string { + t.Helper() + var payloads []string + for _, event := range events { + if event.EventType != "contentBlockStop" { + continue + } + data, err := sonic.Marshal(event.Payload) + if err != nil { + t.Fatalf("marshal contentBlockStop payload: %v", err) + } + payloads = append(payloads, string(data)) + } + return payloads +} + +// TestConverseStreamTextBlockEmitsContentBlockStop replays the stream lifecycle of a +// plain text response and asserts the Converse egress closes the text block with a +// contentBlockStop event before messageStop, as the Bedrock ConverseStream contract +// requires (issue #4262). Text blocks have no contentBlockStart (the ContentBlockStart +// union has no text member), so the block is opened implicitly by its first delta. +func TestConverseStreamTextBlockEmitsContentBlockStop(t *testing.T) { + messageType := schemas.ResponsesMessageTypeMessage + chunks := []*schemas.BifrostResponsesStreamResponse{ + {Type: schemas.ResponsesStreamResponseTypeCreated}, + {Type: schemas.ResponsesStreamResponseTypeInProgress}, + { + Type: schemas.ResponsesStreamResponseTypeOutputItemAdded, + OutputIndex: schemas.Ptr(0), + Item: &schemas.ResponsesMessage{Type: &messageType}, + }, + { + Type: schemas.ResponsesStreamResponseTypeOutputTextDelta, + ContentIndex: schemas.Ptr(0), + Delta: schemas.Ptr("Hello"), + }, + { + Type: schemas.ResponsesStreamResponseTypeOutputTextDelta, + ContentIndex: schemas.Ptr(0), + Delta: schemas.Ptr(" world"), + }, + { + Type: schemas.ResponsesStreamResponseTypeOutputTextDone, + ContentIndex: schemas.Ptr(0), + Text: schemas.Ptr("Hello world"), + }, + { + Type: schemas.ResponsesStreamResponseTypeContentPartDone, + ContentIndex: schemas.Ptr(0), + }, + { + Type: schemas.ResponsesStreamResponseTypeOutputItemDone, + OutputIndex: schemas.Ptr(0), + ContentIndex: schemas.Ptr(0), + Item: &schemas.ResponsesMessage{Type: &messageType}, + }, + { + Type: schemas.ResponsesStreamResponseTypeCompleted, + Response: &schemas.BifrostResponsesResponse{ + Usage: &schemas.ResponsesResponseUsage{InputTokens: 10, OutputTokens: 2, TotalTokens: 12}, + }, + }, + } + + events := encodeConverseStream(t, chunks) + + want := []string{"messageStart", "contentBlockDelta", "contentBlockDelta", "contentBlockStop", "messageStop", "metadata"} + if got := encodedEventTypes(events); !reflect.DeepEqual(got, want) { + t.Fatalf("event sequence mismatch:\n want %v\n got %v", want, got) + } + + stops := contentBlockStopPayloads(t, events) + if len(stops) != 1 || stops[0] != `{"contentBlockIndex":0}` { + t.Errorf(`contentBlockStop payloads: want [{"contentBlockIndex":0}], got %v`, stops) + } +} + +// TestConverseStreamToolUseBlockEmitsContentBlockStop covers a tool use block: the +// contentBlockStart(toolUse) opened at OutputItemAdded must be closed by a +// contentBlockStop carrying the same block index, before messageStop. +func TestConverseStreamToolUseBlockEmitsContentBlockStop(t *testing.T) { + functionCallType := schemas.ResponsesMessageTypeFunctionCall + toolItem := &schemas.ResponsesMessage{ + Type: &functionCallType, + ResponsesToolMessage: &schemas.ResponsesToolMessage{ + CallID: schemas.Ptr("call_1"), + Name: schemas.Ptr("get_weather"), + }, + } + chunks := []*schemas.BifrostResponsesStreamResponse{ + {Type: schemas.ResponsesStreamResponseTypeCreated}, + { + Type: schemas.ResponsesStreamResponseTypeOutputItemAdded, + OutputIndex: schemas.Ptr(1), + ContentIndex: schemas.Ptr(1), + Item: toolItem, + }, + { + Type: schemas.ResponsesStreamResponseTypeFunctionCallArgumentsDelta, + ContentIndex: schemas.Ptr(1), + Delta: schemas.Ptr(`{"location":"Paris"}`), + }, + { + Type: schemas.ResponsesStreamResponseTypeOutputItemDone, + OutputIndex: schemas.Ptr(1), + ContentIndex: schemas.Ptr(1), + Item: toolItem, + }, + { + Type: schemas.ResponsesStreamResponseTypeCompleted, + Response: &schemas.BifrostResponsesResponse{ + Usage: &schemas.ResponsesResponseUsage{InputTokens: 15, OutputTokens: 8, TotalTokens: 23}, + }, + }, + } + + events := encodeConverseStream(t, chunks) + + want := []string{"messageStart", "contentBlockStart", "contentBlockDelta", "contentBlockStop", "messageStop", "metadata"} + if got := encodedEventTypes(events); !reflect.DeepEqual(got, want) { + t.Fatalf("event sequence mismatch:\n want %v\n got %v", want, got) + } + + stops := contentBlockStopPayloads(t, events) + if len(stops) != 1 || stops[0] != `{"contentBlockIndex":1}` { + t.Errorf(`contentBlockStop payloads: want [{"contentBlockIndex":1}], got %v`, stops) + } +} + +// TestConverseStreamReasoningBlockEmitsContentBlockStop covers a reasoning block: +// reasoning deltas stream as contentBlockDelta(reasoningContent) and the block is +// closed by the contentBlockStop emitted on the reasoning item's OutputItemDone. +func TestConverseStreamReasoningBlockEmitsContentBlockStop(t *testing.T) { + reasoningType := schemas.ResponsesMessageTypeReasoning + chunks := []*schemas.BifrostResponsesStreamResponse{ + { + Type: schemas.ResponsesStreamResponseTypeReasoningSummaryTextDelta, + ContentIndex: schemas.Ptr(0), + Delta: schemas.Ptr("thinking"), + }, + { + Type: schemas.ResponsesStreamResponseTypeReasoningSummaryTextDone, + ContentIndex: schemas.Ptr(0), + }, + { + Type: schemas.ResponsesStreamResponseTypeOutputItemDone, + OutputIndex: schemas.Ptr(0), + ContentIndex: schemas.Ptr(0), + Item: &schemas.ResponsesMessage{Type: &reasoningType}, + }, + } + + events := encodeConverseStream(t, chunks) + + want := []string{"contentBlockDelta", "contentBlockStop"} + if got := encodedEventTypes(events); !reflect.DeepEqual(got, want) { + t.Fatalf("event sequence mismatch:\n want %v\n got %v", want, got) + } + + stops := contentBlockStopPayloads(t, events) + if len(stops) != 1 || stops[0] != `{"contentBlockIndex":0}` { + t.Errorf(`contentBlockStop payloads: want [{"contentBlockIndex":0}], got %v`, stops) + } +} diff --git a/core/providers/bedrock/errors.go b/core/providers/bedrock/errors.go index aab50bfa064..f3b176078d0 100644 --- a/core/providers/bedrock/errors.go +++ b/core/providers/bedrock/errors.go @@ -31,18 +31,24 @@ func parseBedrockHTTPError(statusCode int, headers http.Header, body []byte) *sc bifrostErr.Error.Code = errorResp.Code } - if bifrostErr.Type == nil { - exceptionType := errorResp.Type - if exceptionType == "" { - if hv := headers.Get("X-Amzn-Errortype"); hv != "" { - if i := strings.IndexAny(hv, ":#"); i >= 0 { - hv = hv[:i] - } - exceptionType = strings.TrimSpace(hv) + exceptionType := errorResp.Type + if exceptionType == "" { + if hv := headers.Get("X-Amzn-Errortype"); hv != "" { + if i := strings.IndexAny(hv, ":#"); i >= 0 { + hv = hv[:i] } + exceptionType = strings.TrimSpace(hv) } - if exceptionType != "" { - bifrostErr.Type = &exceptionType + } + if exceptionType != "" { + if bifrostErr.Type == nil { + bifrostErr.Type = schemas.Ptr(exceptionType) + } + if bifrostErr.Error == nil { + bifrostErr.Error = &schemas.ErrorField{} + } + if bifrostErr.Error.Type == nil { + bifrostErr.Error.Type = schemas.Ptr(exceptionType) } } diff --git a/core/providers/bedrock/errors_test.go b/core/providers/bedrock/errors_test.go index 1829a2cc9a3..75b30b2fa0d 100644 --- a/core/providers/bedrock/errors_test.go +++ b/core/providers/bedrock/errors_test.go @@ -2,12 +2,45 @@ package bedrock import ( "net/http" + "net/http/httptest" "testing" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) +// TestExecuteBedrockRequest_SurfacesExceptionType is an end-to-end guard for the +// non-streaming executor path (Converse / chat_completion). It must route upstream +// HTTP errors through parseBedrockHTTPError so the AWS exception type — delivered +// here only via the X-Amzn-Errortype header with a ":" qualifier, as real EOL +// models return — is surfaced on both the top-level type and the nested error.type. +// Regression for a bespoke inline error path that dropped the type entirely. +func TestExecuteBedrockRequest_SurfacesExceptionType(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("X-Amzn-Errortype", "ResourceNotFoundException:http://internal.amazon.com/coral/com.amazon.bedrock/") + w.WriteHeader(http.StatusNotFound) + _, _ = w.Write([]byte(`{"message":"This model version has reached the end of its life. Please refer to the AWS documentation for more details."}`)) + })) + defer server.Close() + + req, err := http.NewRequest(http.MethodPost, server.URL, nil) + require.NoError(t, err) + + provider := &BedrockProvider{client: server.Client()} + _, _, _, bifrostErr := provider.executeBedrockRequest(req) + + require.NotNil(t, bifrostErr) + require.NotNil(t, bifrostErr.StatusCode) + assert.Equal(t, http.StatusNotFound, *bifrostErr.StatusCode) + assert.False(t, bifrostErr.IsBifrostError, "non-streaming path delegates retryability to the retry gate via status code") + require.NotNil(t, bifrostErr.Type, "top-level type must be recovered from the header") + assert.Equal(t, "ResourceNotFoundException", *bifrostErr.Type) + require.NotNil(t, bifrostErr.Error) + require.NotNil(t, bifrostErr.Error.Type, "nested error.type must be populated for OpenAI-shaped consumers") + assert.Equal(t, "ResourceNotFoundException", *bifrostErr.Error.Type) + assert.Contains(t, bifrostErr.Error.Message, "reached the end of its life") +} + // TestParseBedrockHTTPError_PreservesExceptionType verifies that the upstream // AWS exception type (the JSON "__type" field) is preserved on the resulting // BifrostError. Without this, a retired/unsupported model error surfaces as a @@ -46,6 +79,40 @@ func TestParseBedrockHTTPError_TypeFromHeader(t *testing.T) { assert.Contains(t, bifrostErr.Error.Message, "reached the end of its life") } +// TestParseBedrockHTTPError_PopulatesNestedErrorType verifies the AWS exception +// type is surfaced on the nested error object (error.type), not only at the +// top level. OpenAI-shaped consumers read error.type/error.code, so a +// Bedrock passthrough error must carry the type there — matching every other +// provider. This is the customer-reported gap: "only error.message, no type". +func TestParseBedrockHTTPError_PopulatesNestedErrorType(t *testing.T) { + headers := http.Header{} + headers.Set("X-Amzn-Errortype", "ValidationException") + body := []byte(`{"message":"Invocation of model ID anthropic.claude-v2 with on-demand throughput isn't supported."}`) + + bifrostErr := parseBedrockHTTPError(http.StatusBadRequest, headers, body) + + require.NotNil(t, bifrostErr) + require.NotNil(t, bifrostErr.Type) + assert.Equal(t, "ValidationException", *bifrostErr.Type, "top-level type must still be set") + require.NotNil(t, bifrostErr.Error) + require.NotNil(t, bifrostErr.Error.Type, "nested error.type must be populated for OpenAI-shaped consumers") + assert.Equal(t, "ValidationException", *bifrostErr.Error.Type) +} + +// TestParseBedrockHTTPError_NestedTypeFromBodyType verifies the nested +// error.type is also populated when the type comes from the body "__type" +// rather than the header. +func TestParseBedrockHTTPError_NestedTypeFromBodyType(t *testing.T) { + body := []byte(`{"__type":"ThrottlingException","message":"rate exceeded"}`) + + bifrostErr := parseBedrockHTTPError(http.StatusTooManyRequests, http.Header{}, body) + + require.NotNil(t, bifrostErr) + require.NotNil(t, bifrostErr.Error) + require.NotNil(t, bifrostErr.Error.Type) + assert.Equal(t, "ThrottlingException", *bifrostErr.Error.Type) +} + // TestParseBedrockHTTPError_HeaderTypeQualifierStripped ensures the trailing // ":" / "#" qualifier AWS sometimes appends to X-Amzn-Errortype is // removed. diff --git a/core/providers/bedrock/mantle.go b/core/providers/bedrock/mantle.go index 45d7b4c10de..30edbbcf04f 100644 --- a/core/providers/bedrock/mantle.go +++ b/core/providers/bedrock/mantle.go @@ -4,7 +4,7 @@ import ( "bytes" "context" "fmt" - "maps" + "io" "net/http" "strings" @@ -13,19 +13,25 @@ import ( schemas "github.com/maximhq/bifrost/core/schemas" ) -// isMantleModel reports whether a model should be routed via the Bedrock Mantle endpoint. -// Accepts "gpt-*"/"gemma-4-*", "openai.gpt-*"/"google.gemma-4-*", or region-prefixed variants. -// Gemma 3 is intentionally excluded: it only supports Chat (not Responses) on mantle, and the -// non-mantle path serves both APIs via Converse, so forcing it to mantle would break Responses. -func isMantleModel(model string) bool { - return strings.Contains(model, "gpt-") || strings.Contains(model, "gemma-4") +// isMantleModel reports whether a model should be routed via the Bedrock Mantle +// OpenAI-compatible endpoint. OpenAI-family (gpt-*) and Gemma 4 models are mantle-only +// (they have no Converse equivalent). Gemma 3 is intentionally excluded: it only supports +// Chat (not Responses) on mantle, and the Converse path serves both APIs, so forcing it to +// mantle would break Responses. +// +// Deprecated: in-provider Bedrock Mantle routing is retained for backwards compatibility. +// New configurations should use the "bedrock_mantle" provider, which owns the Bedrock Mantle +// surface (Claude native-Anthropic, OpenAI-compatible, and Gemma). +func isMantleModel(ctx *schemas.BifrostContext, model string) bool { + return schemas.IsOpenAIModelFamily(ctx, model) || strings.Contains(model, "gemma-4") } -// mantleURL builds the Bedrock Mantle endpoint URL for the given region, model, and API path. -// Pass the canonical (capability-resolved) model for correct path gating; the request body -// still carries the wire request.Model. Frontier families (closed gpt-5.x, Gemma 4) live under -// the "openai/v1" base path; gpt-oss uses the bare "v1" path. -func mantleURL(region, model, path string) string { +// mantleOpenAIURL builds the Bedrock Mantle OpenAI-compatible endpoint URL for the given +// region, model, and API path (e.g. "chat/completions", "responses"). Pass the canonical +// (capability-resolved) model for correct path gating; the request body still carries the +// wire request.Model. Frontier families (closed gpt-5.x, Gemma 4) live under the "openai/v1" +// base path; gpt-oss uses the bare "v1" path. +func mantleOpenAIURL(region, model, path string) string { base := "v1" if strings.Contains(model, "gpt-5") || strings.Contains(model, "gemma-4") { base = "openai/v1" @@ -33,29 +39,66 @@ func mantleURL(region, model, path string) string { return fmt.Sprintf("https://bedrock-mantle.%s.api.aws/%s/%s", region, base, path) } -// mantleSigV4Headers computes SigV4 auth headers for a mantle request by signing a dummy +// SignMantleV4Headers computes SigV4 auth headers for a mantle request by signing a dummy // net/http.Request. jsonData must be the exact bytes that will be sent. accept must match // the Accept header the actual request will send, since SigV4 signs all request headers. -func (provider *BedrockProvider) mantleSigV4Headers( +// extraHeaders are the static headers the actual request will carry; any x-amz-* among them +// are added to the canonical request before signing so the signature covers them (AWS requires +// every x-amz-* header on the wire to be signed, otherwise verification fails). +func SignMantleV4Headers( ctx *schemas.BifrostContext, jsonData []byte, requestURL, accept string, key schemas.Key, region string, + extraHeaders map[string]string, ) (map[string]string, *schemas.BifrostError) { - req, err := http.NewRequestWithContext(ctx, http.MethodPost, requestURL, bytes.NewReader(jsonData)) + method := http.MethodPost + if jsonData == nil { + method = http.MethodGet + } + var body io.Reader + if jsonData != nil { + body = bytes.NewReader(jsonData) + } + + req, err := http.NewRequestWithContext(ctx, method, requestURL, body) if err != nil { return nil, providerUtils.NewBifrostOperationError("failed to create signing request", err) } req.Header.Set("Accept", accept) - if bifrostErr := signAWSRequestFromKey(ctx, req, key.BedrockKeyConfig, region, bedrockMantleSigningService); bifrostErr != nil { + for k, v := range extraHeaders { + if strings.HasPrefix(strings.ToLower(k), "x-amz-") { + req.Header.Set(k, v) + } + } + + keyCfg := key.BedrockKeyConfig + // Create a synthetic BedrockKeyConfig in case of bedrock_mantle: its + // BedrockMantleKeyConfig carries the same credential fields, so map them across + // (Region is passed separately; ARN/batch config are not used for signing). + if key.BedrockMantleKeyConfig != nil && key.BedrockKeyConfig == nil { + keyCfg = &schemas.BedrockKeyConfig{ + AccessKey: key.BedrockMantleKeyConfig.AccessKey, + SecretKey: key.BedrockMantleKeyConfig.SecretKey, + SessionToken: key.BedrockMantleKeyConfig.SessionToken, + RoleARN: key.BedrockMantleKeyConfig.RoleARN, + ExternalID: key.BedrockMantleKeyConfig.ExternalID, + RoleSessionName: key.BedrockMantleKeyConfig.RoleSessionName, + } + } + if bifrostErr := signAWSRequest(ctx, req, keyCfg, region, bedrockMantleSigningService); bifrostErr != nil { return nil, bifrostErr } + // Return the headers exactly as signed: signAWSRequest defaults an empty Accept/Content-Type + // to "application/json" and includes them in SignedHeaders, so the caller must send those same + // values (not the original, possibly-empty, accept) or the signature won't match. headers := map[string]string{ "Authorization": req.Header.Get("Authorization"), "X-Amz-Date": req.Header.Get("X-Amz-Date"), "x-amz-content-sha256": req.Header.Get("x-amz-content-sha256"), - "Accept": accept, + "Accept": req.Header.Get("Accept"), + "Content-Type": req.Header.Get("Content-Type"), } if token := req.Header.Get("X-Amz-Security-Token"); token != "" { headers["X-Amz-Security-Token"] = token @@ -63,33 +106,23 @@ func (provider *BedrockProvider) mantleSigV4Headers( return headers, nil } -// chatCompletionViaMantle handles non-streaming chat completions for mantle (gpt-oss) models. -func (provider *BedrockProvider) chatCompletionViaMantle( +// mantleChatCompletions handles non-streaming chat completions for mantle models (gpt-* +// and Gemma 4) via the Bedrock Mantle OpenAI-compatible endpoint. +func (provider *BedrockProvider) mantleChatCompletions( ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostChatRequest, ) (*schemas.BifrostChatResponse, *schemas.BifrostError) { region := resolveBedrockRegion(ctx, key, request.Model) - url := mantleURL(region, schemas.ResolveCanonicalModel(ctx, request.Model), "chat/completions") + url := mantleOpenAIURL(region, schemas.ResolveCanonicalModel(ctx, request.Model), "chat/completions") - // Build extraHeaders: always start with network-config headers, then overlay SigV4 if needed. - // Allocate explicitly so maps.Copy never writes into a nil map. - extraHeaders := make(map[string]string, len(provider.networkConfig.ExtraHeaders)) - maps.Copy(extraHeaders, provider.networkConfig.ExtraHeaders) + // SigV4 (empty key value): sign the exact body the handler builds via a signer closure. + // Bearer (key has a value): no signer; auth flows through the Authorization header. + var signer providerUtils.BodySigner if key.Value.GetValue() == "" { - // SigV4: pre-build body for signing. HandleOpenAIChatCompletionRequest rebuilds the - // same bytes (deterministic marshaling), so the signature stays valid. - jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody(ctx, request, func() (providerUtils.RequestBodyWithExtraParams, error) { - return openai.ToOpenAIChatRequest(ctx, request), nil - }) - if bifrostErr != nil { - return nil, bifrostErr + signer = func(body []byte) (map[string]string, *schemas.BifrostError) { + return SignMantleV4Headers(ctx, body, url, "application/json", key, region, provider.networkConfig.ExtraHeaders) } - sigHeaders, bifrostErr := provider.mantleSigV4Headers(ctx, jsonData, url, "application/json", key, region) - if bifrostErr != nil { - return nil, bifrostErr - } - maps.Copy(extraHeaders, sigHeaders) } return openai.HandleOpenAIChatCompletionRequest( @@ -97,18 +130,21 @@ func (provider *BedrockProvider) chatCompletionViaMantle( provider.mantleClient, url, request, - key, - extraHeaders, + openai.BearerAuthHeader(key), + provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), - nil, nil, + nil, + nil, + signer, provider.logger, ) } -// chatCompletionStreamViaMantle handles streaming chat completions for mantle (gpt-oss) models. -func (provider *BedrockProvider) chatCompletionStreamViaMantle( +// mantleChatCompletionsStream handles streaming chat completions for mantle models (gpt-* +// and Gemma 4) via the Bedrock Mantle OpenAI-compatible endpoint. +func (provider *BedrockProvider) mantleChatCompletionsStream( ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), @@ -116,78 +152,52 @@ func (provider *BedrockProvider) chatCompletionStreamViaMantle( request *schemas.BifrostChatRequest, ) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { region := resolveBedrockRegion(ctx, key, request.Model) - url := mantleURL(region, schemas.ResolveCanonicalModel(ctx, request.Model), "chat/completions") - - // Bearer: identical to Groq / any OpenAI-compatible provider. - if key.Value.GetValue() != "" { - authHeader := map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - return openai.HandleOpenAIChatCompletionStreaming( - ctx, provider.mantleStreamingClient, url, request, - authHeader, provider.networkConfig.ExtraHeaders, - provider.networkConfig.StreamIdleTimeoutInSeconds, - providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), - providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), - provider.GetProviderKey(), postHookRunner, - nil, nil, nil, nil, nil, - provider.logger, postHookSpanFinalizer, - ) - } + url := mantleOpenAIURL(region, schemas.ResolveCanonicalModel(ctx, request.Model), "chat/completions") - // SigV4: pre-build body to sign, then pass it via customRequestConverter so the handler - // sends the exact same bytes we signed. - openaiReq := openai.ToOpenAIChatRequest(ctx, request) - openaiReq.Stream = schemas.Ptr(true) - openaiReq.StreamOptions = &schemas.ChatStreamOptions{IncludeUsage: schemas.Ptr(true)} - - jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody(ctx, request, func() (providerUtils.RequestBodyWithExtraParams, error) { - return openaiReq, nil - }) - if bifrostErr != nil { - return nil, bifrostErr - } - authHeader, bifrostErr := provider.mantleSigV4Headers(ctx, jsonData, url, "text/event-stream", key, region) - if bifrostErr != nil { - return nil, bifrostErr + // SigV4 (empty key value): sign the exact body the handler builds via a signer closure. + // Bearer (key has a value): no signer; auth flows through the Authorization header. + var signer providerUtils.BodySigner + if key.Value.GetValue() == "" { + signer = func(body []byte) (map[string]string, *schemas.BifrostError) { + return SignMantleV4Headers(ctx, body, url, "text/event-stream", key, region, provider.networkConfig.ExtraHeaders) + } } return openai.HandleOpenAIChatCompletionStreaming( ctx, provider.mantleStreamingClient, url, request, - authHeader, provider.networkConfig.ExtraHeaders, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), postHookRunner, - func(_ *schemas.BifrostChatRequest) (providerUtils.RequestBodyWithExtraParams, error) { - return openaiReq, nil - }, - nil, nil, nil, nil, - provider.logger, postHookSpanFinalizer, + nil, + nil, + nil, + nil, + nil, + signer, + provider.logger, + postHookSpanFinalizer, ) } -// responsesViaMantle handles non-streaming Responses API requests for mantle (gpt-oss) models. -func (provider *BedrockProvider) responsesViaMantle( +// mantleResponses handles non-streaming Responses API requests for mantle models (gpt-* +// and Gemma 4) via the Bedrock Mantle OpenAI-compatible endpoint. +func (provider *BedrockProvider) mantleResponses( ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostResponsesRequest, ) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) { region := resolveBedrockRegion(ctx, key, request.Model) - url := mantleURL(region, schemas.ResolveCanonicalModel(ctx, request.Model), "responses") + url := mantleOpenAIURL(region, schemas.ResolveCanonicalModel(ctx, request.Model), "responses") - extraHeaders := make(map[string]string, len(provider.networkConfig.ExtraHeaders)) - maps.Copy(extraHeaders, provider.networkConfig.ExtraHeaders) + // SigV4 (empty key value): sign the exact body the handler builds via a signer closure. + // Bearer (key has a value): no signer; auth flows through the Authorization header. + var signer providerUtils.BodySigner if key.Value.GetValue() == "" { - jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody(ctx, request, func() (providerUtils.RequestBodyWithExtraParams, error) { - return openai.ToOpenAIResponsesRequest(ctx, request), nil - }) - if bifrostErr != nil { - return nil, bifrostErr + signer = func(body []byte) (map[string]string, *schemas.BifrostError) { + return SignMantleV4Headers(ctx, body, url, "application/json", key, region, provider.networkConfig.ExtraHeaders) } - sigHeaders, bifrostErr := provider.mantleSigV4Headers(ctx, jsonData, url, "application/json", key, region) - if bifrostErr != nil { - return nil, bifrostErr - } - maps.Copy(extraHeaders, sigHeaders) } return openai.HandleOpenAIResponsesRequest( @@ -195,18 +205,21 @@ func (provider *BedrockProvider) responsesViaMantle( provider.mantleClient, url, request, - key, - extraHeaders, + openai.BearerAuthHeader(key), + provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), - nil, nil, + nil, + nil, + signer, provider.logger, ) } -// responsesStreamViaMantle handles streaming Responses API requests for mantle (gpt-oss) models. -func (provider *BedrockProvider) responsesStreamViaMantle( +// mantleResponsesStream handles streaming Responses API requests for mantle models (gpt-* +// and Gemma 4) via the Bedrock Mantle OpenAI-compatible endpoint. +func (provider *BedrockProvider) mantleResponsesStream( ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), @@ -214,50 +227,30 @@ func (provider *BedrockProvider) responsesStreamViaMantle( request *schemas.BifrostResponsesRequest, ) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { region := resolveBedrockRegion(ctx, key, request.Model) - url := mantleURL(region, schemas.ResolveCanonicalModel(ctx, request.Model), "responses") - - // Bearer: identical to Groq / any OpenAI-compatible provider. - if key.Value.GetValue() != "" { - authHeader := map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - return openai.HandleOpenAIResponsesStreaming( - ctx, provider.mantleStreamingClient, url, request, - authHeader, provider.networkConfig.ExtraHeaders, - provider.networkConfig.StreamIdleTimeoutInSeconds, - providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), - providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), - provider.GetProviderKey(), postHookRunner, - nil, nil, nil, nil, - provider.logger, postHookSpanFinalizer, - ) - } + url := mantleOpenAIURL(region, schemas.ResolveCanonicalModel(ctx, request.Model), "responses") - // SigV4: pre-build body to sign. - openaiReq := openai.ToOpenAIResponsesRequest(ctx, request) - openaiReq.Stream = schemas.Ptr(true) - - jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody(ctx, request, func() (providerUtils.RequestBodyWithExtraParams, error) { - return openaiReq, nil - }) - if bifrostErr != nil { - return nil, bifrostErr - } - authHeader, bifrostErr := provider.mantleSigV4Headers(ctx, jsonData, url, "text/event-stream", key, region) - if bifrostErr != nil { - return nil, bifrostErr + // SigV4 (empty key value): sign the exact body the handler builds via a signer closure. + // Bearer (key has a value): no signer; auth flows through the Authorization header. + var signer providerUtils.BodySigner + if key.Value.GetValue() == "" { + signer = func(body []byte) (map[string]string, *schemas.BifrostError) { + return SignMantleV4Headers(ctx, body, url, "text/event-stream", key, region, provider.networkConfig.ExtraHeaders) + } } return openai.HandleOpenAIResponsesStreaming( ctx, provider.mantleStreamingClient, url, request, - authHeader, provider.networkConfig.ExtraHeaders, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), postHookRunner, - nil, nil, - func(_ *openai.OpenAIResponsesRequest) *openai.OpenAIResponsesRequest { - return openaiReq - }, nil, - provider.logger, postHookSpanFinalizer, + nil, + nil, + nil, + signer, + provider.logger, + postHookSpanFinalizer, ) } diff --git a/core/providers/bedrock/mantle_test.go b/core/providers/bedrock/mantle_test.go index 129fb643f97..11e2bc04f88 100644 --- a/core/providers/bedrock/mantle_test.go +++ b/core/providers/bedrock/mantle_test.go @@ -1,8 +1,14 @@ package bedrock -import "testing" +import ( + "context" + "testing" + + schemas "github.com/maximhq/bifrost/core/schemas" +) func TestIsMantleModel(t *testing.T) { + ctx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline) cases := []struct { model string want bool @@ -24,19 +30,20 @@ func TestIsMantleModel(t *testing.T) { {"gemma-3-12b-it", false}, {"google.gemma-3-27b-it", false}, {"gemma-3-4b-it", false}, - // other families stay on the Converse path + // Anthropic (Claude) models stay on the Converse path. {"claude-opus-4-8", false}, {"anthropic.claude-3-5-sonnet-20240620-v1:0", false}, + // other families stay on the Converse path {"amazon.titan-text-express-v1", false}, } for _, tc := range cases { - if got := isMantleModel(tc.model); got != tc.want { + if got := isMantleModel(ctx, tc.model); got != tc.want { t.Errorf("isMantleModel(%q) = %v, want %v", tc.model, got, tc.want) } } } -func TestMantleURL(t *testing.T) { +func TestMantleOpenAIURL(t *testing.T) { cases := []struct { name string region string @@ -57,8 +64,8 @@ func TestMantleURL(t *testing.T) { } for _, tc := range cases { t.Run(tc.name, func(t *testing.T) { - if got := mantleURL(tc.region, tc.model, tc.path); got != tc.want { - t.Errorf("mantleURL(%q, %q, %q) = %q, want %q", tc.region, tc.model, tc.path, got, tc.want) + if got := mantleOpenAIURL(tc.region, tc.model, tc.path); got != tc.want { + t.Errorf("mantleOpenAIURL(%q, %q, %q) = %q, want %q", tc.region, tc.model, tc.path, got, tc.want) } }) } diff --git a/core/providers/bedrock/responses.go b/core/providers/bedrock/responses.go index d3f3f47d6bf..020ba979c43 100644 --- a/core/providers/bedrock/responses.go +++ b/core/providers/bedrock/responses.go @@ -1510,8 +1510,27 @@ func FinalizeBedrockStream(state *BedrockResponsesStreamState, sequenceNumber in } } + // Set Status/IncompleteDetails and the terminal event type per OpenAI's + // Responses-API contract, matching the non-streaming switch above so + // unmapped reasons leave Status unset on both paths. + terminalEventType := schemas.ResponsesStreamResponseTypeCompleted + if response.StopReason != nil { + switch *response.StopReason { + case string(schemas.BifrostFinishReasonLength): + terminalEventType = schemas.ResponsesStreamResponseTypeIncomplete + response.Status = schemas.Ptr(schemas.ResponsesResponseStatusIncomplete) + response.IncompleteDetails = &schemas.ResponsesResponseIncompleteDetails{ + Reason: schemas.ResponsesResponseIncompleteReasonMaxOutputTokens, + } + case string(schemas.BifrostFinishReasonStop), string(schemas.BifrostFinishReasonToolCalls): + if response.Status == nil { + response.Status = schemas.Ptr(schemas.ResponsesResponseStatusCompleted) + } + } + } + responses = append(responses, &schemas.BifrostResponsesStreamResponse{ - Type: schemas.ResponsesStreamResponseTypeCompleted, + Type: terminalEventType, SequenceNumber: sequenceNumber + len(responses), Response: response, }) @@ -1754,12 +1773,20 @@ func ToBedrockConverseStreamResponse(bifrostResp *schemas.BifrostResponsesStream case schemas.ResponsesStreamResponseTypeOutputTextDone, schemas.ResponsesStreamResponseTypeContentPartDone, schemas.ResponsesStreamResponseTypeReasoningSummaryTextDone: - // Content block done - Bedrock doesn't have explicit done events, so we skip them + // Content block done - the contentBlockStop is emitted on OutputItemDone, + // matching the invoke path return nil, nil case schemas.ResponsesStreamResponseTypeOutputItemDone: - // Item done - Bedrock doesn't have explicit done events, so we skip them - return nil, nil + // Item done - emit contentBlockStop. Bedrock terminates every content block + // with a contentBlockStop event carrying the block's index; consumers that + // assemble the message on block boundaries never finalize a block without it. + contentBlockIndex := 0 + if bifrostResp.ContentIndex != nil { + contentBlockIndex = *bifrostResp.ContentIndex + } + event.ContentBlockIndex = &contentBlockIndex + event.ContentBlockStop = true case schemas.ResponsesStreamResponseTypeCompleted: // Message stop - always set stopReason @@ -1851,6 +1878,17 @@ func (event *BedrockStreamEvent) ToEncodedEvents() []BedrockEncodedEvent { }) } + if event.ContentBlockStop { + events = append(events, BedrockEncodedEvent{ + EventType: "contentBlockStop", + Payload: struct { + ContentBlockIndex *int `json:"contentBlockIndex"` + }{ + ContentBlockIndex: event.ContentBlockIndex, + }, + }) + } + if event.StopReason != nil { events = append(events, BedrockEncodedEvent{ EventType: "messageStop", @@ -2696,6 +2734,19 @@ func (response *BedrockConverseResponse) ToBifrostResponsesResponse(ctx *schemas } } bifrostResp.StopReason = &stopReason + // Surface truncation via Status + IncompleteDetails per OpenAI's + // Responses-API contract; without these, truncations are silent. + switch stopReason { + case string(schemas.BifrostFinishReasonLength): + bifrostResp.Status = schemas.Ptr(schemas.ResponsesResponseStatusIncomplete) + bifrostResp.IncompleteDetails = &schemas.ResponsesResponseIncompleteDetails{ + Reason: schemas.ResponsesResponseIncompleteReasonMaxOutputTokens, + } + case string(schemas.BifrostFinishReasonStop), string(schemas.BifrostFinishReasonToolCalls): + if bifrostResp.Status == nil { + bifrostResp.Status = schemas.Ptr(schemas.ResponsesResponseStatusCompleted) + } + } } if response.Trace != nil { diff --git a/core/providers/bedrock/streambuffering_test.go b/core/providers/bedrock/streambuffering_test.go index 51fce483438..3c11a7ad921 100644 --- a/core/providers/bedrock/streambuffering_test.go +++ b/core/providers/bedrock/streambuffering_test.go @@ -98,7 +98,9 @@ func TestChatCompletionStream_StreamsIncrementally_NotBuffered(t *testing.T) { ctx := testBedrockCtx() key := testBedrockKey() - streamChan, bifrostErr := provider.ChatCompletionStream(ctx, noopPostHookRunner, nil, key, testChatRequest()) + req := testChatRequest() + req.Model = testConverseStreamModel // Converse eventstream path (OpenAI-family models stream via Mantle; Anthropic uses Converse) + streamChan, bifrostErr := provider.ChatCompletionStream(ctx, noopPostHookRunner, nil, key, req) require.Nil(t, bifrostErr, "stream setup should not error") require.NotNil(t, streamChan) @@ -147,7 +149,9 @@ func TestMakeStreamingRequest_SendsIdentityAcceptEncoding(t *testing.T) { ctx := testBedrockCtx() key := testBedrockKey() - streamChan, bifrostErr := provider.ChatCompletionStream(ctx, noopPostHookRunner, nil, key, testChatRequest()) + req := testChatRequest() + req.Model = testConverseStreamModel // Converse eventstream path (OpenAI-family models stream via Mantle; Anthropic uses Converse) + streamChan, bifrostErr := provider.ChatCompletionStream(ctx, noopPostHookRunner, nil, key, req) require.Nil(t, bifrostErr) require.NotNil(t, streamChan) for range streamChan { // drain so the request fully completes diff --git a/core/providers/bedrock/transport_test.go b/core/providers/bedrock/transport_test.go index 28311569215..4fed985eb4d 100644 --- a/core/providers/bedrock/transport_test.go +++ b/core/providers/bedrock/transport_test.go @@ -108,6 +108,12 @@ func noopPostHookRunner(_ *schemas.BifrostContext, result *schemas.BifrostRespon return result, err } +// testConverseStreamModel is a non-Anthropic, non-OpenAI Bedrock model that routes through +// the Converse streaming path (and thus the AWS EventStream decoder). OpenAI-family models +// stream via the Mantle endpoint instead; Anthropic/Claude also route through Converse, but +// Nova is used here to keep the EventStream-exception cases provider-agnostic. +const testConverseStreamModel = "amazon.nova-lite-v1:0" + // testChatRequest returns a minimal BifrostChatRequest for streaming tests. func testChatRequest() *schemas.BifrostChatRequest { content := "hello" @@ -331,7 +337,9 @@ func TestChatCompletionStream_RetryableException_ChunkIsRetryable(t *testing.T) ctx := testBedrockCtx() key := testBedrockKey() - streamChan, bifrostErr := provider.ChatCompletionStream(ctx, noopPostHookRunner, nil, key, testChatRequest()) + req := testChatRequest() + req.Model = testConverseStreamModel + streamChan, bifrostErr := provider.ChatCompletionStream(ctx, noopPostHookRunner, nil, key, req) require.Nil(t, bifrostErr, "expected EventStream exception to surface as a stream chunk") require.NotNil(t, streamChan) @@ -378,7 +386,9 @@ func TestChatCompletionStream_NonRetryableException_IsTerminal(t *testing.T) { ctx := testBedrockCtx() key := testBedrockKey() - streamChan, bifrostErr := provider.ChatCompletionStream(ctx, noopPostHookRunner, nil, key, testChatRequest()) + req := testChatRequest() + req.Model = testConverseStreamModel + streamChan, bifrostErr := provider.ChatCompletionStream(ctx, noopPostHookRunner, nil, key, req) require.Nil(t, bifrostErr, "expected EventStream exception to surface as a stream chunk") require.NotNil(t, streamChan) @@ -517,7 +527,9 @@ func TestResponsesStream_RetryableException_ChunkIsRetryable(t *testing.T) { defer ts.Close() provider := newTestProviderWithServer(t, ts) - streamChan, bifrostErr := provider.ResponsesStream(testBedrockCtx(), noopPostHookRunner, nil, testBedrockKey(), testResponsesRequest()) + req := testResponsesRequest() + req.Model = testConverseStreamModel + streamChan, bifrostErr := provider.ResponsesStream(testBedrockCtx(), noopPostHookRunner, nil, testBedrockKey(), req) assertRetryableExceptionChunk(t, streamChan, bifrostErr, tc.excType, tc.expectedStatus) }) } diff --git a/core/providers/bedrock/types.go b/core/providers/bedrock/types.go index 1ba6682bcf1..bc605253a01 100644 --- a/core/providers/bedrock/types.go +++ b/core/providers/bedrock/types.go @@ -761,6 +761,11 @@ type BedrockStreamEvent struct { // Start field for tool use events Start *BedrockContentBlockStart `json:"start,omitempty"` // For contentBlockStart events + // Marker for contentBlockStop events. The wire payload of contentBlockStop carries + // only contentBlockIndex, so the flat union needs an explicit flag to represent it. + // Never serialized directly: ToEncodedEvents builds the payload from ContentBlockIndex. + ContentBlockStop bool `json:"-"` + // Metadata and usage (can appear at top level) Usage *BedrockTokenUsage `json:"usage,omitempty"` // Usage information Metrics *BedrockConverseMetrics `json:"metrics,omitempty"` // Performance metrics diff --git a/core/providers/bedrockmantle/bedrockmantle.go b/core/providers/bedrockmantle/bedrockmantle.go new file mode 100644 index 00000000000..78ab092ba5f --- /dev/null +++ b/core/providers/bedrockmantle/bedrockmantle.go @@ -0,0 +1,617 @@ +// Package bedrockmantle implements the Bedrock Mantle LLM provider. It owns the Bedrock Mantle +// surface served on the bedrock-mantle.{region}.api.aws host: Claude models via the native +// Anthropic Messages API (/anthropic/v1/messages), and OpenAI-family (gpt-*) and Gemma models +// via the OpenAI-compatible API (/v1 or /openai/v1). The model id is sent verbatim, save for an +// optional leading "region/" addressing prefix. Authentication is either a Bedrock Mantle API key +// (Authorization: Bearer) or AWS SigV4 for the bedrock-mantle service. +package bedrockmantle + +import ( + "context" + "fmt" + "maps" + "strings" + "time" + + "github.com/maximhq/bifrost/core/providers/anthropic" + "github.com/maximhq/bifrost/core/providers/bedrock" + openai "github.com/maximhq/bifrost/core/providers/openai" + providerUtils "github.com/maximhq/bifrost/core/providers/utils" + schemas "github.com/maximhq/bifrost/core/schemas" + "github.com/valyala/fasthttp" +) + +// BedrockMantleProvider implements the Provider interface for the Bedrock Mantle endpoint. +type BedrockMantleProvider struct { + logger schemas.Logger // Logger for provider operations + mantleClient *fasthttp.Client // fasthttp client for unary requests (OpenAI-compatible and native-Anthropic paths) + mantleStreamingClient *fasthttp.Client // fasthttp streaming client for streaming requests + networkConfig schemas.NetworkConfig // Network configuration including extra headers + sendBackRawRequest bool // Whether to include raw request in BifrostResponse + sendBackRawResponse bool // Whether to include raw response in BifrostResponse +} + +// NewBedrockMantleProvider creates a new Bedrock Mantle provider instance. +// It initializes the fasthttp unary and streaming clients with the provided configuration. +// There is no default BaseURL: mantle requests target computed bedrock-mantle.{region}.api.aws hosts. +func NewBedrockMantleProvider(config *schemas.ProviderConfig, logger schemas.Logger) (*BedrockMantleProvider, error) { + config.CheckAndSetDefaults() + + requestTimeout := time.Second * time.Duration(config.NetworkConfig.DefaultRequestTimeoutInSeconds) + + // fasthttp clients for Bedrock Mantle (shared by OpenAI-compatible and native-Anthropic paths). + // ReadTimeout is the shared provider request timeout; oversized Anthropic responses are handled + // by PrepareResponseStreaming, not by these static settings. + mantleFasthttpClient := &fasthttp.Client{ + ReadTimeout: requestTimeout, + WriteTimeout: requestTimeout, + MaxConnsPerHost: config.NetworkConfig.MaxConnsPerHost, + MaxIdleConnDuration: 30 * time.Second, + MaxConnWaitTimeout: requestTimeout, + MaxConnDuration: time.Second * time.Duration(schemas.DefaultMaxConnDurationInSeconds), + ConnPoolStrategy: fasthttp.FIFO, + } + mantleFasthttpClient = providerUtils.ConfigureProxy(mantleFasthttpClient, config.ProxyConfig, logger) + mantleFasthttpClient = providerUtils.ConfigureDialer(mantleFasthttpClient, config.NetworkConfig.AllowPrivateNetwork) + mantleFasthttpClient = providerUtils.ConfigureTLS(mantleFasthttpClient, config.NetworkConfig, logger) + mantleStreamingFasthttpClient := providerUtils.BuildStreamingClient(mantleFasthttpClient) + + return &BedrockMantleProvider{ + logger: logger, + mantleClient: mantleFasthttpClient, + mantleStreamingClient: mantleStreamingFasthttpClient, + networkConfig: config.NetworkConfig, + sendBackRawRequest: config.SendBackRawRequest, + sendBackRawResponse: config.SendBackRawResponse, + }, nil +} + +// mantleAnthropicVersion is the Anthropic API version sent as an HTTP header on the +// Bedrock Mantle native-Anthropic endpoint (unlike bedrock-runtime, which carries the +// version as an "anthropic_version" body field). +const mantleAnthropicVersion = "2023-06-01" + +// defaultMantleRegion is the fallback AWS region used to build the bedrock-mantle host when +// no region is supplied by the model prefix, the resolved alias, or the key config. +const defaultMantleRegion = "us-east-1" + +// mantleOpenAIURL builds the Bedrock Mantle OpenAI-compatible endpoint URL for the given +// region, model, and API path (e.g. "chat/completions", "responses"). The native-Anthropic +// path is built separately by mantleAnthropicURL. Pass the canonical (capability-resolved) +// model for correct path gating; the request body still carries the wire request.Model. +// Frontier families (closed gpt-5.x, Gemma 4) live under the "openai/v1" base path; gpt-oss +// uses the bare "v1" path. +func mantleOpenAIURL(region, model, path string) string { + base := "v1" + if strings.Contains(model, "gpt-5") || strings.Contains(model, "gemma-4") { + base = "openai/v1" + } + return fmt.Sprintf("https://bedrock-mantle.%s.api.aws/%s/%s", region, base, path) +} + +// mantleAnthropicURL builds the Bedrock Mantle native-Anthropic Messages endpoint URL. +func mantleAnthropicURL(region string) string { + return fmt.Sprintf("https://bedrock-mantle.%s.api.aws/anthropic/v1/messages", region) +} + +// mantleSigner returns a BodySigner that SigV4-signs the request body for the bedrock-mantle +// service, or nil when a Bedrock Mantle API key is present (auth then flows through the +// Authorization: Bearer header instead). The handler invokes the signer on the exact body it +// builds, so the signature always covers what is actually sent. +func (provider *BedrockMantleProvider) mantleSigner(ctx *schemas.BifrostContext, key schemas.Key, url, accept, region string) providerUtils.BodySigner { + if key.Value.GetValue() != "" { + return nil + } + return func(body []byte) (map[string]string, *schemas.BifrostError) { + return bedrock.SignMantleV4Headers(ctx, body, url, accept, key, region, provider.networkConfig.ExtraHeaders) + } +} + +// GetProviderKey returns the provider identifier for Bedrock Mantle. +func (provider *BedrockMantleProvider) GetProviderKey() schemas.ModelProvider { + return schemas.BedrockMantle +} + +// listModelsByKey lists models from the Bedrock Mantle (OpenAI-compatible) /v1/models +// endpoint for a single key, converted to a Bifrost response with the key's allow/blacklist/ +// alias gating. The request is signed as it is sent (the GET cannot reuse the POST signer); +// a Bedrock Mantle API key authenticates via Authorization: Bearer, otherwise the request is +// SigV4-signed for the bedrock-mantle service and the signed headers are merged into the +// per-request extra headers consumed by the shared OpenAI list-models path. +func (provider *BedrockMantleProvider) listModelsByKey(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostListModelsRequest) (*schemas.BifrostListModelsResponse, *schemas.BifrostError) { + region := provider.resolveRegion(ctx, key, "") + mURL := mantleOpenAIURL(region, "", "models") + + extraHeaders := provider.networkConfig.ExtraHeaders + if key.Value.GetValue() == "" { + // SigV4: sign the GET and overlay the signed headers; OpenAI's ListModelsByKey only sets + // a Bearer header when the key carries a value, so the SigV4 Authorization wins here. + sigHeaders, bifrostErr := bedrock.SignMantleV4Headers(ctx, nil, mURL, "", key, region, provider.networkConfig.ExtraHeaders) + if bifrostErr != nil { + return nil, bifrostErr + } + merged := make(map[string]string, len(provider.networkConfig.ExtraHeaders)+len(sigHeaders)) + maps.Copy(merged, provider.networkConfig.ExtraHeaders) + maps.Copy(merged, sigHeaders) + extraHeaders = merged + } + + return openai.ListModelsByKey( + ctx, + provider.mantleClient, + mURL, + key, + request.Unfiltered, + extraHeaders, + provider.GetProviderKey(), + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + ) +} + +// ListModels lists models from the Bedrock Mantle OpenAI-compatible /v1/models endpoint, +// aggregating across the supplied keys. +func (provider *BedrockMantleProvider) ListModels(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostListModelsRequest) (*schemas.BifrostListModelsResponse, *schemas.BifrostError) { + return providerUtils.HandleMultipleListModelsRequests( + ctx, + keys, + request, + provider.listModelsByKey, + ) +} + +// ChatCompletion performs a chat completion request to the Bedrock Mantle endpoint, dispatching +// by model family: Anthropic-family (Claude) models use the native Anthropic Messages surface; +// all other (OpenAI-family / Gemma) models use the OpenAI-compatible surface. +func (provider *BedrockMantleProvider) ChatCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostChatRequest) (*schemas.BifrostChatResponse, *schemas.BifrostError) { + region := provider.resolveRegion(ctx, key, request.Model) + + // Anthropic-family models (Claude) use the native Anthropic Messages surface; all other + // (OpenAI-family / Gemma) models use the OpenAI-compatible surface. + if schemas.IsAnthropicModelFamily(ctx, request.Model) { + url := mantleAnthropicURL(region) + _, bareModel := parseBedrockRegionAndModel(request.Model) + return anthropic.HandleAnthropicChatCompletionRequest( + ctx, + provider.mantleClient, + url, + request, + anthropic.AnthropicRequestBuildConfig{ + Provider: schemas.BedrockMantle, + Model: bareModel, + BetaHeaderOverrides: provider.networkConfig.BetaHeaderOverrides, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }, + openai.BearerAuthHeader(key), + addAnthropicHeaders(provider.networkConfig.ExtraHeaders), + provider.mantleSigner(ctx, key, url, "application/json", region), + provider.logger, + ) + } + + url := mantleOpenAIURL(region, schemas.ResolveCanonicalModel(ctx, request.Model), "chat/completions") + return openai.HandleOpenAIChatCompletionRequest( + ctx, + provider.mantleClient, + url, + request, + openai.BearerAuthHeader(key), + provider.networkConfig.ExtraHeaders, + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + provider.GetProviderKey(), + nil, + nil, + provider.mantleSigner(ctx, key, url, "application/json", region), + provider.logger, + ) +} + +// ChatCompletionStream performs a streaming chat completion request to the Bedrock Mantle +// endpoint, dispatching by model family (native Anthropic vs OpenAI-compatible). +func (provider *BedrockMantleProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + region := provider.resolveRegion(ctx, key, request.Model) + + // Anthropic-family models (Claude) use the native Anthropic Messages surface; all other + // (OpenAI-family / Gemma) models use the OpenAI-compatible surface. + if schemas.IsAnthropicModelFamily(ctx, request.Model) { + url := mantleAnthropicURL(region) + + _, bareModel := parseBedrockRegionAndModel(request.Model) + jsonData, bifrostErr := anthropic.BuildAnthropicChatRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{ + Provider: schemas.BedrockMantle, + Model: bareModel, + IsStreaming: true, + BetaHeaderOverrides: provider.networkConfig.BetaHeaderOverrides, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }) + if bifrostErr != nil { + return nil, bifrostErr + } + + return anthropic.HandleAnthropicChatCompletionStreaming( + ctx, + provider.mantleStreamingClient, + url, + jsonData, + openai.BearerAuthHeader(key), + addAnthropicHeaders(provider.networkConfig.ExtraHeaders), + provider.networkConfig.StreamIdleTimeoutInSeconds, + provider.networkConfig.BetaHeaderOverrides, + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + provider.GetProviderKey(), + postHookRunner, + nil, + provider.mantleSigner(ctx, key, url, "text/event-stream", region), + provider.logger, + postHookSpanFinalizer, + ) + } + + url := mantleOpenAIURL(region, schemas.ResolveCanonicalModel(ctx, request.Model), "chat/completions") + return openai.HandleOpenAIChatCompletionStreaming( + ctx, provider.mantleStreamingClient, url, request, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, + provider.networkConfig.StreamIdleTimeoutInSeconds, + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + provider.GetProviderKey(), postHookRunner, + nil, nil, nil, nil, nil, + provider.mantleSigner(ctx, key, url, "text/event-stream", region), + provider.logger, postHookSpanFinalizer, + ) +} + +// Responses performs a Responses API request to the Bedrock Mantle endpoint, dispatching by +// model family (native Anthropic vs OpenAI-compatible). +func (provider *BedrockMantleProvider) Responses(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostResponsesRequest) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) { + region := provider.resolveRegion(ctx, key, request.Model) + + // Anthropic-family models (Claude) use the native Anthropic Messages surface; all other + // (OpenAI-family / Gemma) models use the OpenAI-compatible surface. + if schemas.IsAnthropicModelFamily(ctx, request.Model) { + url := mantleAnthropicURL(region) + + _, bareModel := parseBedrockRegionAndModel(request.Model) + return anthropic.HandleAnthropicResponsesRequest( + ctx, + provider.mantleClient, + url, + request, + anthropic.AnthropicRequestBuildConfig{ + Provider: schemas.BedrockMantle, + Model: bareModel, + ValidateTools: true, + BetaHeaderOverrides: provider.networkConfig.BetaHeaderOverrides, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }, + openai.BearerAuthHeader(key), + addAnthropicHeaders(provider.networkConfig.ExtraHeaders), + provider.mantleSigner(ctx, key, url, "application/json", region), + provider.logger, + ) + } + + url := mantleOpenAIURL(region, schemas.ResolveCanonicalModel(ctx, request.Model), "responses") + return openai.HandleOpenAIResponsesRequest( + ctx, + provider.mantleClient, + url, + request, + openai.BearerAuthHeader(key), + provider.networkConfig.ExtraHeaders, + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + provider.GetProviderKey(), + nil, nil, + provider.mantleSigner(ctx, key, url, "application/json", region), + provider.logger, + ) +} + +// ResponsesStream performs a streaming Responses API request to the Bedrock Mantle endpoint, +// dispatching by model family (native Anthropic vs OpenAI-compatible). +func (provider *BedrockMantleProvider) ResponsesStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostResponsesRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + region := provider.resolveRegion(ctx, key, request.Model) + + // Anthropic-family models (Claude) use the native Anthropic Messages surface; all other + // (OpenAI-family / Gemma) models use the OpenAI-compatible surface. + if schemas.IsAnthropicModelFamily(ctx, request.Model) { + url := mantleAnthropicURL(region) + + _, bareModel := parseBedrockRegionAndModel(request.Model) + jsonData, bifrostErr := anthropic.BuildAnthropicResponsesRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{ + Provider: schemas.BedrockMantle, + Model: bareModel, + IsStreaming: true, + ValidateTools: true, + BetaHeaderOverrides: provider.networkConfig.BetaHeaderOverrides, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }) + if bifrostErr != nil { + return nil, bifrostErr + } + + return anthropic.HandleAnthropicResponsesStream( + ctx, + provider.mantleStreamingClient, + url, + jsonData, + openai.BearerAuthHeader(key), + addAnthropicHeaders(provider.networkConfig.ExtraHeaders), + provider.networkConfig.StreamIdleTimeoutInSeconds, + provider.networkConfig.BetaHeaderOverrides, + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + provider.GetProviderKey(), + postHookRunner, + nil, + provider.mantleSigner(ctx, key, url, "text/event-stream", region), + provider.logger, + postHookSpanFinalizer, + ) + } + + url := mantleOpenAIURL(region, schemas.ResolveCanonicalModel(ctx, request.Model), "responses") + return openai.HandleOpenAIResponsesStreaming( + ctx, provider.mantleStreamingClient, url, request, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, + provider.networkConfig.StreamIdleTimeoutInSeconds, + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + provider.GetProviderKey(), postHookRunner, + nil, nil, nil, nil, + provider.mantleSigner(ctx, key, url, "text/event-stream", region), + provider.logger, postHookSpanFinalizer, + ) +} + +// TextCompletion is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) TextCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostTextCompletionRequest) (*schemas.BifrostTextCompletionResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.TextCompletionRequest, provider.GetProviderKey()) +} + +// TextCompletionStream is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) TextCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostTextCompletionRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.TextCompletionStreamRequest, provider.GetProviderKey()) +} + +// Embedding is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) Embedding(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostEmbeddingRequest) (*schemas.BifrostEmbeddingResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.EmbeddingRequest, provider.GetProviderKey()) +} + +// Speech is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) Speech(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostSpeechRequest) (*schemas.BifrostSpeechResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.SpeechRequest, provider.GetProviderKey()) +} + +// Rerank is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) Rerank(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostRerankRequest) (*schemas.BifrostRerankResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.RerankRequest, provider.GetProviderKey()) +} + +// OCR is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) OCR(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostOCRRequest) (*schemas.BifrostOCRResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.OCRRequest, provider.GetProviderKey()) +} + +// SpeechStream is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) SpeechStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostSpeechRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.SpeechStreamRequest, provider.GetProviderKey()) +} + +// Transcription is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) Transcription(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostTranscriptionRequest) (*schemas.BifrostTranscriptionResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.TranscriptionRequest, provider.GetProviderKey()) +} + +// TranscriptionStream is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) TranscriptionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostTranscriptionRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.TranscriptionStreamRequest, provider.GetProviderKey()) +} + +// ImageGeneration is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) ImageGeneration(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostImageGenerationRequest) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageGenerationRequest, provider.GetProviderKey()) +} + +// ImageGenerationStream is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) ImageGenerationStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostImageGenerationRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageGenerationStreamRequest, provider.GetProviderKey()) +} + +// ImageEdit is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) ImageEdit(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostImageEditRequest) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageEditRequest, provider.GetProviderKey()) +} + +// ImageEditStream is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) ImageEditStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostImageEditRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageEditStreamRequest, provider.GetProviderKey()) +} + +// ImageVariation is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) ImageVariation(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostImageVariationRequest) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageVariationRequest, provider.GetProviderKey()) +} + +// VideoGeneration is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) VideoGeneration(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoGenerationRequest) (*schemas.BifrostVideoGenerationResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoGenerationRequest, provider.GetProviderKey()) +} + +// VideoRetrieve is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) VideoRetrieve(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoRetrieveRequest) (*schemas.BifrostVideoGenerationResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoRetrieveRequest, provider.GetProviderKey()) +} + +// VideoDownload is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) VideoDownload(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoDownloadRequest) (*schemas.BifrostVideoDownloadResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoDownloadRequest, provider.GetProviderKey()) +} + +// VideoDelete is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) VideoDelete(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoDeleteRequest) (*schemas.BifrostVideoDeleteResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoDeleteRequest, provider.GetProviderKey()) +} + +// VideoList is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) VideoList(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoListRequest) (*schemas.BifrostVideoListResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoListRequest, provider.GetProviderKey()) +} + +// VideoRemix is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) VideoRemix(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoRemixRequest) (*schemas.BifrostVideoGenerationResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoRemixRequest, provider.GetProviderKey()) +} + +// FileUpload is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) FileUpload(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostFileUploadRequest) (*schemas.BifrostFileUploadResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.FileUploadRequest, provider.GetProviderKey()) +} + +// FileList is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) FileList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileListRequest) (*schemas.BifrostFileListResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.FileListRequest, provider.GetProviderKey()) +} + +// FileRetrieve is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) FileRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileRetrieveRequest) (*schemas.BifrostFileRetrieveResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.FileRetrieveRequest, provider.GetProviderKey()) +} + +// FileDelete is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) FileDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileDeleteRequest) (*schemas.BifrostFileDeleteResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.FileDeleteRequest, provider.GetProviderKey()) +} + +// FileContent is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) FileContent(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileContentRequest) (*schemas.BifrostFileContentResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.FileContentRequest, provider.GetProviderKey()) +} + +// BatchCreate is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) BatchCreate(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostBatchCreateRequest) (*schemas.BifrostBatchCreateResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchCreateRequest, provider.GetProviderKey()) +} + +// BatchList is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) BatchList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchListRequest) (*schemas.BifrostBatchListResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchListRequest, provider.GetProviderKey()) +} + +// BatchRetrieve is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) BatchRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchRetrieveRequest) (*schemas.BifrostBatchRetrieveResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchRetrieveRequest, provider.GetProviderKey()) +} + +// BatchCancel is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) BatchCancel(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchCancelRequest) (*schemas.BifrostBatchCancelResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchCancelRequest, provider.GetProviderKey()) +} + +// BatchDelete is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) BatchDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchDeleteRequest) (*schemas.BifrostBatchDeleteResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchDeleteRequest, provider.GetProviderKey()) +} + +// BatchResults is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) BatchResults(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchResultsRequest) (*schemas.BifrostBatchResultsResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchResultsRequest, provider.GetProviderKey()) +} + +// CountTokens is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) CountTokens(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostResponsesRequest) (*schemas.BifrostCountTokensResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.CountTokensRequest, provider.GetProviderKey()) +} + +// Compaction is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) Compaction(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostCompactionRequest) (*schemas.BifrostCompactionResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.CompactionRequest, provider.GetProviderKey()) +} + +// CachedContentCreate is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) CachedContentCreate(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostCachedContentCreateRequest) (*schemas.BifrostCachedContentCreateResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentCreateRequest, provider.GetProviderKey()) +} + +// CachedContentList is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) CachedContentList(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentListRequest) (*schemas.BifrostCachedContentListResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentListRequest, provider.GetProviderKey()) +} + +// CachedContentRetrieve is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) CachedContentRetrieve(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentRetrieveRequest) (*schemas.BifrostCachedContentRetrieveResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentRetrieveRequest, provider.GetProviderKey()) +} + +// CachedContentUpdate is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) CachedContentUpdate(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentUpdateRequest) (*schemas.BifrostCachedContentUpdateResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentUpdateRequest, provider.GetProviderKey()) +} + +// CachedContentDelete is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) CachedContentDelete(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentDeleteRequest) (*schemas.BifrostCachedContentDeleteResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentDeleteRequest, provider.GetProviderKey()) +} + +// ContainerCreate is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) ContainerCreate(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostContainerCreateRequest) (*schemas.BifrostContainerCreateResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerCreateRequest, provider.GetProviderKey()) +} + +// ContainerList is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) ContainerList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerListRequest) (*schemas.BifrostContainerListResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerListRequest, provider.GetProviderKey()) +} + +// ContainerRetrieve is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) ContainerRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerRetrieveRequest) (*schemas.BifrostContainerRetrieveResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerRetrieveRequest, provider.GetProviderKey()) +} + +// ContainerDelete is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) ContainerDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerDeleteRequest) (*schemas.BifrostContainerDeleteResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerDeleteRequest, provider.GetProviderKey()) +} + +// ContainerFileCreate is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) ContainerFileCreate(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostContainerFileCreateRequest) (*schemas.BifrostContainerFileCreateResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileCreateRequest, provider.GetProviderKey()) +} + +// ContainerFileList is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) ContainerFileList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileListRequest) (*schemas.BifrostContainerFileListResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileListRequest, provider.GetProviderKey()) +} + +// ContainerFileRetrieve is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) ContainerFileRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileRetrieveRequest) (*schemas.BifrostContainerFileRetrieveResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileRetrieveRequest, provider.GetProviderKey()) +} + +// ContainerFileContent is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) ContainerFileContent(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileContentRequest) (*schemas.BifrostContainerFileContentResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileContentRequest, provider.GetProviderKey()) +} + +// ContainerFileDelete is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) ContainerFileDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileDeleteRequest) (*schemas.BifrostContainerFileDeleteResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileDeleteRequest, provider.GetProviderKey()) +} + +// Passthrough is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) Passthrough(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostPassthroughRequest) (*schemas.BifrostPassthroughResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.PassthroughRequest, provider.GetProviderKey()) +} + +// PassthroughStream is not supported by the Bedrock Mantle provider. +func (provider *BedrockMantleProvider) PassthroughStream(_ *schemas.BifrostContext, _ schemas.PostHookRunner, _ func(context.Context), _ schemas.Key, _ *schemas.BifrostPassthroughRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.PassthroughStreamRequest, provider.GetProviderKey()) +} diff --git a/core/providers/bedrockmantle/bedrockmantle_test.go b/core/providers/bedrockmantle/bedrockmantle_test.go new file mode 100644 index 00000000000..431ca0a2a58 --- /dev/null +++ b/core/providers/bedrockmantle/bedrockmantle_test.go @@ -0,0 +1,93 @@ +package bedrockmantle_test + +import ( + "os" + "strings" + "testing" + + "github.com/maximhq/bifrost/core/internal/llmtests" + schemas "github.com/maximhq/bifrost/core/schemas" +) + +// TestBedrockMantle runs the comprehensive harness against the bedrock_mantle provider. +// +// It is gated on AWS credentials (the SigV4 path needs them). The Claude scenarios exercise the +// native-Anthropic Messages surface; gpt-oss exercises the OpenAI-compatible surface. Only the +// operations the provider actually implements are enabled — everything else (embeddings, rerank, +// batch, files, image edit/variation, count tokens, text completion) is an unsupported stub. +// +// The model ids live in the harness account (GetKeysForProvider, case BedrockMantle); if a model +// is reported as not found, tune the aliases there to whatever the mantle endpoints accept. +func TestBedrockMantle(t *testing.T) { + t.Parallel() + + if strings.TrimSpace(os.Getenv("AWS_ACCESS_KEY_ID")) == "" || strings.TrimSpace(os.Getenv("AWS_SECRET_ACCESS_KEY")) == "" { + t.Skip("Skipping Bedrock Mantle tests because AWS_ACCESS_KEY_ID or AWS_SECRET_ACCESS_KEY is not set") + } + + client, ctx, cancel, err := llmtests.SetupTest() + if err != nil { + t.Fatalf("Error initializing test setup: %v", err) + } + defer cancel() + defer client.Shutdown() + + testConfig := llmtests.ComprehensiveTestConfig{ + Provider: schemas.BedrockMantle, + ChatModel: "anthropic.claude-haiku-4-5", + PromptCachingModel: "anthropic.claude-opus-4-8", + VisionModel: "anthropic.claude-haiku-4-5", + Fallbacks: []schemas.Fallback{ + {Provider: schemas.BedrockMantle, Model: "anthropic.claude-opus-4-8"}, + }, + ReasoningModel: "anthropic.claude-opus-4-8", + InterleavedThinkingModel: "anthropic.claude-opus-4-8", + Scenarios: llmtests.TestScenarios{ + // Supported: chat + responses surfaces (native-Anthropic and OpenAI-compatible). + SimpleChat: true, + CompletionStream: true, + MultiTurnConversation: true, + ToolCalls: true, + ToolCallsStreaming: true, + MultipleToolCalls: true, + MultipleToolCallsStreaming: true, + End2EndToolCalling: true, + AutomaticFunctionCall: true, + ImageBase64: true, // Claude vision (native-Anthropic) + CompleteEnd2End: true, + ListModels: true, + Reasoning: true, + InterleavedThinking: true, + EagerInputStreaming: true, + StructuredOutputs: true, + PromptCaching: true, + + // Unsupported by the mantle provider (unsupported-operation stubs). + TextCompletion: false, + ImageURL: false, // native-Anthropic does not accept image URLs + MultipleImages: false, + FileBase64: false, + FileURL: false, + Embedding: false, + Rerank: false, + BatchCreate: false, + BatchList: false, + BatchRetrieve: false, + BatchCancel: false, + BatchResults: false, + FileUpload: false, + FileList: false, + FileRetrieve: false, + FileDelete: false, + FileContent: false, + FileBatchInput: false, + CountTokens: false, + ImageEdit: false, + ImageVariation: false, + }, + } + + t.Run("BedrockMantleTests", func(t *testing.T) { + llmtests.RunAllComprehensiveTests(t, client, ctx, testConfig) + }) +} diff --git a/core/providers/bedrockmantle/utils.go b/core/providers/bedrockmantle/utils.go new file mode 100644 index 00000000000..d49998cdbc9 --- /dev/null +++ b/core/providers/bedrockmantle/utils.go @@ -0,0 +1,60 @@ +package bedrockmantle + +import ( + "maps" + "regexp" + "strings" + + schemas "github.com/maximhq/bifrost/core/schemas" +) + +// awsRegionRegex matches valid AWS region identifiers (e.g. "us-east-1", "eu-north-1", "us-gov-east-1"). +// (?:-[a-z]+)+ allows multi-segment directional parts so GovCloud regions (us-gov-east-1) are +// recognised alongside standard single-segment ones (eu-north-1, ap-southeast-2). +var awsRegionRegex = regexp.MustCompile(`^[a-z]{2,3}(?:-[a-z]+)+-\d+$`) + +// addAnthropicHeaders returns a copy of the given extra headers with the native-Anthropic +// mantle anthropic-version header added. It clones the input first so it never mutates the +// shared networkConfig.ExtraHeaders map (which would leak anthropic-version onto the +// OpenAI-compatible requests and race across concurrent requests). +func addAnthropicHeaders(headers map[string]string) map[string]string { + out := maps.Clone(headers) + if out == nil { + out = make(map[string]string, 1) + } + out["anthropic-version"] = mantleAnthropicVersion + return out +} + +// parseBedrockRegionAndModel splits a model string that optionally carries an AWS region prefix +// into its region and bare model ID components. +// If no region prefix is present the returned region is empty and bareModel equals model. +func parseBedrockRegionAndModel(model string) (region, bareModel string) { + if idx := strings.IndexByte(model, '/'); idx > 0 { + prefix := model[:idx] + if awsRegionRegex.MatchString(prefix) { + return prefix, model[idx+1:] + } + } + return "", model +} + +// resolveRegion returns the AWS region to use for a request. +// Priority: model-string region prefix > alias-level Region > key-level +// BedrockMantleKeyConfig.Region > defaultMantleRegion. The model-string prefix +// stays highest since it's the most explicit signal — when an admin types a +// region into their model ID they expect that to win. +func (provider *BedrockMantleProvider) resolveRegion(ctx *schemas.BifrostContext, key schemas.Key, model string) string { + if region, _ := parseBedrockRegionAndModel(model); region != "" { + return region + } + if ra := schemas.GetResolvedAlias(ctx); ra != nil && ra.Config != nil && ra.Config.Region != nil { + if v := ra.Config.Region.GetValue(); v != "" { + return v + } + } + if key.BedrockMantleKeyConfig != nil && key.BedrockMantleKeyConfig.Region != nil && key.BedrockMantleKeyConfig.Region.GetValue() != "" { + return key.BedrockMantleKeyConfig.Region.GetValue() + } + return defaultMantleRegion +} diff --git a/core/providers/cerebras/cerebras.go b/core/providers/cerebras/cerebras.go index b6a3a4d5f4c..b589a7678ca 100644 --- a/core/providers/cerebras/cerebras.go +++ b/core/providers/cerebras/cerebras.go @@ -89,7 +89,7 @@ func (provider *CerebrasProvider) TextCompletion(ctx *schemas.BifrostContext, ke provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -104,17 +104,12 @@ func (provider *CerebrasProvider) TextCompletion(ctx *schemas.BifrostContext, ke // It formats the request, sends it to Cerebras, and processes the response. // Returns a channel of BifrostStreamChunk objects or an error if the request fails. func (provider *CerebrasProvider) TextCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostTextCompletionRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } - // Use shared OpenAI-compatible streaming logic return openai.HandleOpenAITextCompletionStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+"/v1/completions", request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -136,13 +131,14 @@ func (provider *CerebrasProvider) ChatCompletion(ctx *schemas.BifrostContext, ke provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), nil, nil, + nil, provider.logger, ) } @@ -152,17 +148,12 @@ func (provider *CerebrasProvider) ChatCompletion(ctx *schemas.BifrostContext, ke // Uses Cerebras's OpenAI-compatible streaming format. // Returns a channel containing BifrostStreamChunk objects representing the stream or an error if the request fails. func (provider *CerebrasProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } - // Use shared OpenAI-compatible streaming logic return openai.HandleOpenAIChatCompletionStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+"/v1/chat/completions", request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -174,6 +165,7 @@ func (provider *CerebrasProvider) ChatCompletionStream(ctx *schemas.BifrostConte nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) diff --git a/core/providers/cohere/cohere.go b/core/providers/cohere/cohere.go index 92a89a74320..d21df7ef2eb 100644 --- a/core/providers/cohere/cohere.go +++ b/core/providers/cohere/cohere.go @@ -203,7 +203,7 @@ func (provider *CohereProvider) completeRequest(ctx *schemas.BifrostContext, jso // Handle error response if resp.StatusCode() != fasthttp.StatusOK { providerUtils.MaterializeStreamErrorBody(ctx, resp) - return nil, latency, providerResponseHeaders, parseCohereError(resp) + return nil, latency, providerResponseHeaders, providerUtils.SetErrorLatency(parseCohereError(resp), latency) } body, isLargeResp, decodeErr := providerUtils.FinalizeResponseWithLargeDetection(ctx, resp, provider.logger) @@ -271,7 +271,7 @@ func (provider *CohereProvider) listModelsByKey(ctx *schemas.BifrostContext, key // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, parseCohereError(resp) + return nil, providerUtils.SetErrorLatency(parseCohereError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -361,7 +361,7 @@ func (provider *CohereProvider) ChatCompletion(ctx *schemas.BifrostContext, key ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Large response mode: return lightweight response with metadata only @@ -381,7 +381,7 @@ func (provider *CohereProvider) ChatCompletion(ctx *schemas.BifrostContext, key rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, response, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse := response.ToBifrostChatResponse(request.Model) @@ -457,6 +457,7 @@ func (provider *CohereProvider) ChatCompletionStream(ctx *schemas.BifrostContext startTime := time.Now() // Make the request err := provider.streamingClient.Do(req, resp) + latency := time.Since(startTime) if usedLargePayloadBody { providerUtils.DrainLargePayloadRemainder(ctx) } @@ -470,16 +471,16 @@ func (provider *CohereProvider) ChatCompletionStream(ctx *schemas.BifrostContext Message: schemas.ErrRequestCancelled, Error: err, }, - }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Request failed before the first response byte (server closed an idle/pooled connection, // broken pipe, connection refused, DNS failure, etc.). Surface as a retriable upstream // connection error (502) so executeRequestWithRetries honors max_retries, matching the // non-streaming path - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Extract provider response headers before status check so error responses also forward them @@ -488,9 +489,11 @@ func (provider *CohereProvider) ChatCompletionStream(ctx *schemas.BifrostContext // Check for HTTP errors if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) - return nil, providerUtils.EnrichError(ctx, parseCohereError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseCohereError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } + providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) + // Large payload streaming passthrough — pipe raw upstream SSE to client if providerUtils.SetupStreamingPassthrough(ctx, resp) { responseChan := make(chan *schemas.BifrostStreamChunk) @@ -501,8 +504,6 @@ func (provider *CohereProvider) ChatCompletionStream(ctx *schemas.BifrostContext // Create response channel responseChan := make(chan *schemas.BifrostStreamChunk, schemas.DefaultStreamBufferSize) - providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) - // Start streaming in a goroutine go func() { defer providerUtils.EnsureStreamFinalizerCalled(ctx, postHookSpanFinalizer) @@ -640,7 +641,7 @@ func (provider *CohereProvider) Responses(ctx *schemas.BifrostContext, key schem ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Large response mode: return lightweight response with metadata only @@ -660,7 +661,7 @@ func (provider *CohereProvider) Responses(ctx *schemas.BifrostContext, key schem rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, response, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse := response.ToBifrostResponsesResponse() @@ -737,6 +738,7 @@ func (provider *CohereProvider) ResponsesStream(ctx *schemas.BifrostContext, pos startTime := time.Now() // Make the request err := provider.streamingClient.Do(req, resp) + latency := time.Since(startTime) if usedLargePayloadBody { providerUtils.DrainLargePayloadRemainder(ctx) } @@ -750,16 +752,16 @@ func (provider *CohereProvider) ResponsesStream(ctx *schemas.BifrostContext, pos Message: schemas.ErrRequestCancelled, Error: err, }, - }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Request failed before the first response byte (server closed an idle/pooled connection, // broken pipe, connection refused, DNS failure, etc.). Surface as a retriable upstream // connection error (502) so executeRequestWithRetries honors max_retries, matching the // non-streaming path - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Extract provider response headers before status check so error responses also forward them @@ -768,9 +770,11 @@ func (provider *CohereProvider) ResponsesStream(ctx *schemas.BifrostContext, pos // Check for HTTP errors if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) - return nil, providerUtils.EnrichError(ctx, parseCohereError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseCohereError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } + providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) + // Large payload streaming passthrough — pipe raw upstream SSE to client if providerUtils.SetupStreamingPassthrough(ctx, resp) { responseChan := make(chan *schemas.BifrostStreamChunk) @@ -781,8 +785,6 @@ func (provider *CohereProvider) ResponsesStream(ctx *schemas.BifrostContext, pos // Create response channel responseChan := make(chan *schemas.BifrostStreamChunk, schemas.DefaultStreamBufferSize) - providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) - // Start streaming in a goroutine go func() { defer providerUtils.EnsureStreamFinalizerCalled(ctx, postHookSpanFinalizer) @@ -917,7 +919,7 @@ func (provider *CohereProvider) Embedding(ctx *schemas.BifrostContext, key schem ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Large response mode: return lightweight response with metadata only @@ -937,7 +939,7 @@ func (provider *CohereProvider) Embedding(ctx *schemas.BifrostContext, key schem rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, response, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse := response.ToBifrostEmbeddingResponse() @@ -981,7 +983,7 @@ func (provider *CohereProvider) Rerank(ctx *schemas.BifrostContext, key schemas. ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Large response mode: return lightweight response with metadata only @@ -1001,7 +1003,7 @@ func (provider *CohereProvider) Rerank(ctx *schemas.BifrostContext, key schemas. rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, response, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } returnDocuments := request.Params != nil && request.Params.ReturnDocuments != nil && *request.Params.ReturnDocuments @@ -1186,7 +1188,7 @@ func (provider *CohereProvider) CountTokens(ctx *schemas.BifrostContext, key sch ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Large response mode: return lightweight response with metadata only @@ -1210,12 +1212,12 @@ func (provider *CohereProvider) CountTokens(ctx *schemas.BifrostContext, key sch providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), ) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse := cohereResponse.ToBifrostCountTokensResponse(request.Model) if bifrostResponse == nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, fmt.Errorf("nil cohere count tokens response")), jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, fmt.Errorf("nil cohere count tokens response")), jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse.ExtraFields.Latency = latency.Milliseconds() diff --git a/core/providers/cohere/errors.go b/core/providers/cohere/errors.go index e444d866509..311e346bbdb 100644 --- a/core/providers/cohere/errors.go +++ b/core/providers/cohere/errors.go @@ -17,5 +17,9 @@ func parseCohereError(resp *fasthttp.Response) *schemas.BifrostError { if errorResp.Code != nil { bifrostErr.Error.Code = errorResp.Code } + if errorResp.Type != "" && bifrostErr.Error.Type == nil { + typeCopy := errorResp.Type + bifrostErr.Error.Type = &typeCopy + } return bifrostErr } diff --git a/core/providers/cohere/errors_test.go b/core/providers/cohere/errors_test.go new file mode 100644 index 00000000000..5f7270b46b0 --- /dev/null +++ b/core/providers/cohere/errors_test.go @@ -0,0 +1,28 @@ +package cohere + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "github.com/valyala/fasthttp" +) + +// TestParseCohereError_PopulatesNestedErrorType verifies the upstream exception +// type is surfaced on the nested error object (error.type), not only at the top +// level, so OpenAI-shaped consumers see it. +func TestParseCohereError_PopulatesNestedErrorType(t *testing.T) { + var resp fasthttp.Response + resp.SetStatusCode(fasthttp.StatusBadRequest) + resp.SetBodyString(`{"type":"invalid_request_error","message":"model not found"}`) + + bifrostErr := parseCohereError(&resp) + + require.NotNil(t, bifrostErr) + require.NotNil(t, bifrostErr.Error) + require.NotNil(t, bifrostErr.Error.Type, "nested error.type must be populated") + assert.Equal(t, "invalid_request_error", *bifrostErr.Error.Type) + require.NotNil(t, bifrostErr.Type, "top-level type must remain populated") + assert.Equal(t, "invalid_request_error", *bifrostErr.Type) + assert.Equal(t, "model not found", bifrostErr.Error.Message) +} diff --git a/core/providers/deepseek/cachedcontents.go b/core/providers/deepseek/cachedcontents.go new file mode 100644 index 00000000000..8aeff19e65a --- /dev/null +++ b/core/providers/deepseek/cachedcontents.go @@ -0,0 +1,34 @@ +package deepseek + +import ( + providerUtils "github.com/maximhq/bifrost/core/providers/utils" + "github.com/maximhq/bifrost/core/schemas" +) + +// CachedContentCreate is unsupported on DeepSeekProvider. Only Gemini and Vertex AI +// implement the cached-content lifecycle (Google AI Studio + Vertex AI named +// caches). Other providers either lack named cache management entirely or +// handle caching implicitly via per-message cache_control markers. +func (provider *DeepSeekProvider) CachedContentCreate(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostCachedContentCreateRequest) (*schemas.BifrostCachedContentCreateResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentCreateRequest, provider.GetProviderKey()) +} + +// CachedContentList is unsupported on DeepSeekProvider (see CachedContentCreate). +func (provider *DeepSeekProvider) CachedContentList(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentListRequest) (*schemas.BifrostCachedContentListResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentListRequest, provider.GetProviderKey()) +} + +// CachedContentRetrieve is unsupported on DeepSeekProvider (see CachedContentCreate). +func (provider *DeepSeekProvider) CachedContentRetrieve(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentRetrieveRequest) (*schemas.BifrostCachedContentRetrieveResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentRetrieveRequest, provider.GetProviderKey()) +} + +// CachedContentUpdate is unsupported on DeepSeekProvider (see CachedContentCreate). +func (provider *DeepSeekProvider) CachedContentUpdate(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentUpdateRequest) (*schemas.BifrostCachedContentUpdateResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentUpdateRequest, provider.GetProviderKey()) +} + +// CachedContentDelete is unsupported on DeepSeekProvider (see CachedContentCreate). +func (provider *DeepSeekProvider) CachedContentDelete(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostCachedContentDeleteRequest) (*schemas.BifrostCachedContentDeleteResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.CachedContentDeleteRequest, provider.GetProviderKey()) +} diff --git a/core/providers/deepseek/deepseek.go b/core/providers/deepseek/deepseek.go new file mode 100644 index 00000000000..1ea6bae06d9 --- /dev/null +++ b/core/providers/deepseek/deepseek.go @@ -0,0 +1,408 @@ +// Package deepseek implements the DeepSeek LLM provider. +package deepseek + +import ( + "context" + "strings" + "time" + + "github.com/maximhq/bifrost/core/providers/openai" + providerUtils "github.com/maximhq/bifrost/core/providers/utils" + schemas "github.com/maximhq/bifrost/core/schemas" + "github.com/valyala/fasthttp" +) + +// DeepSeekProvider implements the Provider interface for DeepSeek's API. +type DeepSeekProvider struct { + logger schemas.Logger // Logger for provider operations + client *fasthttp.Client // HTTP client for unary API requests (ReadTimeout bounds overall response) + streamingClient *fasthttp.Client // HTTP client for streaming API requests (no ReadTimeout; idle governed by NewIdleTimeoutReader) + networkConfig schemas.NetworkConfig // Network configuration including extra headers + sendBackRawRequest bool // Whether to include raw request in BifrostResponse + sendBackRawResponse bool // Whether to include raw response in BifrostResponse +} + +// NewDeepSeekProvider creates a new DeepSeek provider instance. +// It initializes the HTTP client with the provided configuration and sets up response pools. +// The client is configured with timeouts, concurrency limits, and optional proxy settings. +func NewDeepSeekProvider(config *schemas.ProviderConfig, logger schemas.Logger) (*DeepSeekProvider, error) { + config.CheckAndSetDefaults() + + requestTimeout := time.Second * time.Duration(config.NetworkConfig.DefaultRequestTimeoutInSeconds) + client := &fasthttp.Client{ + ReadTimeout: requestTimeout, + WriteTimeout: requestTimeout, + MaxConnsPerHost: config.NetworkConfig.MaxConnsPerHost, + MaxIdleConnDuration: 30 * time.Second, + MaxConnWaitTimeout: requestTimeout, + MaxConnDuration: time.Second * time.Duration(schemas.DefaultMaxConnDurationInSeconds), + ConnPoolStrategy: fasthttp.FIFO, + } + + // Configure proxy and retry policy + client = providerUtils.ConfigureProxy(client, config.ProxyConfig, logger) + client = providerUtils.ConfigureDialer(client, config.NetworkConfig.AllowPrivateNetwork) + client = providerUtils.ConfigureTLS(client, config.NetworkConfig, logger) + streamingClient := providerUtils.BuildStreamingClient(client) + // Set default BaseURL if not provided + if config.NetworkConfig.BaseURL == "" { + config.NetworkConfig.BaseURL = "https://api.deepseek.com" + } + config.NetworkConfig.BaseURL = strings.TrimRight(config.NetworkConfig.BaseURL, "/") + + return &DeepSeekProvider{ + logger: logger, + client: client, + streamingClient: streamingClient, + networkConfig: config.NetworkConfig, + sendBackRawRequest: config.SendBackRawRequest, + sendBackRawResponse: config.SendBackRawResponse, + }, nil +} + +// GetProviderKey returns the provider identifier for DeepSeek. +func (provider *DeepSeekProvider) GetProviderKey() schemas.ModelProvider { + return schemas.DeepSeek +} + +// ListModels performs a list models request to DeepSeek's API. +func (provider *DeepSeekProvider) ListModels(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostListModelsRequest) (*schemas.BifrostListModelsResponse, *schemas.BifrostError) { + return openai.HandleOpenAIListModelsRequest( + ctx, + provider.client, + request, + provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/models"), + keys, + provider.networkConfig.ExtraHeaders, + provider.GetProviderKey(), + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + ) +} + +// TextCompletion performs a text completion request to DeepSeek's API. +// It formats the request, sends it to DeepSeek, and processes the response. +// Returns a BifrostResponse containing the completion results or an error if the request fails. +func (provider *DeepSeekProvider) TextCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostTextCompletionRequest) (*schemas.BifrostTextCompletionResponse, *schemas.BifrostError) { + ctx.SetValue(schemas.BifrostContextKeyPassthroughExtraParams, true) + return openai.HandleOpenAITextCompletionRequest( + ctx, + provider.client, + provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/beta/completions"), + request, + openai.BearerAuthHeader(key), + provider.networkConfig.ExtraHeaders, + provider.GetProviderKey(), + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + nil, + nil, + provider.logger, + ) +} + +// TextCompletionStream performs a streaming text completion request to DeepSeek's API. +// It formats the request, sends it to DeepSeek, and processes the response. +// Returns a channel of BifrostStreamChunk objects or an error if the request fails. +func (provider *DeepSeekProvider) TextCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostTextCompletionRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + ctx.SetValue(schemas.BifrostContextKeyPassthroughExtraParams, true) + return openai.HandleOpenAITextCompletionStreaming( + ctx, + provider.streamingClient, + provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/beta/completions"), + request, + openai.BearerAuthHeader(key), + provider.networkConfig.ExtraHeaders, + provider.networkConfig.StreamIdleTimeoutInSeconds, + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + provider.GetProviderKey(), + nil, + postHookRunner, + nil, + nil, + provider.logger, + postHookSpanFinalizer, + ) +} + +// ChatCompletion performs a chat completion request to the DeepSeek API. +func (provider *DeepSeekProvider) ChatCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostChatRequest) (*schemas.BifrostChatResponse, *schemas.BifrostError) { + ctx.SetValue(schemas.BifrostContextKeyPassthroughExtraParams, true) + return openai.HandleOpenAIChatCompletionRequest( + ctx, + provider.client, + provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/chat/completions"), + request, + openai.BearerAuthHeader(key), + provider.networkConfig.ExtraHeaders, + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + provider.GetProviderKey(), + nil, + nil, + nil, + provider.logger, + ) +} + +// ChatCompletionStream performs a streaming chat completion request to the DeepSeek API. +// It supports real-time streaming of responses using Server-Sent Events (SSE). +// Uses DeepSeek's OpenAI-compatible streaming format. +// Returns a channel containing BifrostStreamChunk objects representing the stream or an error if the request fails. +func (provider *DeepSeekProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + ctx.SetValue(schemas.BifrostContextKeyPassthroughExtraParams, true) + return openai.HandleOpenAIChatCompletionStreaming( + ctx, + provider.streamingClient, + provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/chat/completions"), + request, + openai.BearerAuthHeader(key), + provider.networkConfig.ExtraHeaders, + provider.networkConfig.StreamIdleTimeoutInSeconds, + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + schemas.DeepSeek, + postHookRunner, + nil, + nil, + nil, + nil, + nil, + nil, + provider.logger, + postHookSpanFinalizer, + ) +} + +func (provider *DeepSeekProvider) Responses(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostResponsesRequest) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) { + chatResponse, err := provider.ChatCompletion(ctx, key, request.ToChatRequest()) + if err != nil { + return nil, err + } + + response := chatResponse.ToBifrostResponsesResponse() + + return response, nil +} + +// ResponsesStream performs a streaming responses request to the DeepSeek API. +func (provider *DeepSeekProvider) ResponsesStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostResponsesRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + ctx.SetValue(schemas.BifrostContextKeyIsResponsesToChatCompletionFallback, true) + return provider.ChatCompletionStream( + ctx, + postHookRunner, + postHookSpanFinalizer, + key, + request.ToChatRequest(), + ) +} + +// Embedding is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) Embedding(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostEmbeddingRequest) (*schemas.BifrostEmbeddingResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.EmbeddingRequest, provider.GetProviderKey()) +} + +// Speech is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) Speech(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostSpeechRequest) (*schemas.BifrostSpeechResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.SpeechRequest, provider.GetProviderKey()) +} + +// Rerank is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) Rerank(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostRerankRequest) (*schemas.BifrostRerankResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.RerankRequest, provider.GetProviderKey()) +} + +// OCR is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) OCR(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostOCRRequest) (*schemas.BifrostOCRResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.OCRRequest, provider.GetProviderKey()) +} + +// SpeechStream is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) SpeechStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostSpeechRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.SpeechStreamRequest, provider.GetProviderKey()) +} + +// Transcription is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) Transcription(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostTranscriptionRequest) (*schemas.BifrostTranscriptionResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.TranscriptionRequest, provider.GetProviderKey()) +} + +// TranscriptionStream is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) TranscriptionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostTranscriptionRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.TranscriptionStreamRequest, provider.GetProviderKey()) +} + +// ImageGeneration is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) ImageGeneration(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostImageGenerationRequest) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageGenerationRequest, provider.GetProviderKey()) +} + +// ImageGenerationStream is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) ImageGenerationStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostImageGenerationRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageGenerationStreamRequest, provider.GetProviderKey()) +} + +// ImageEdit is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) ImageEdit(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostImageEditRequest) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageEditRequest, provider.GetProviderKey()) +} + +// ImageEditStream is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) ImageEditStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostImageEditRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageEditStreamRequest, provider.GetProviderKey()) +} + +// ImageVariation is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) ImageVariation(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostImageVariationRequest) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ImageVariationRequest, provider.GetProviderKey()) +} + +// VideoGeneration is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) VideoGeneration(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoGenerationRequest) (*schemas.BifrostVideoGenerationResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoGenerationRequest, provider.GetProviderKey()) +} + +// VideoRetrieve is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) VideoRetrieve(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoRetrieveRequest) (*schemas.BifrostVideoGenerationResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoRetrieveRequest, provider.GetProviderKey()) +} + +// VideoDownload is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) VideoDownload(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoDownloadRequest) (*schemas.BifrostVideoDownloadResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoDownloadRequest, provider.GetProviderKey()) +} + +// VideoDelete is not supported by DeepSeek provider. +func (provider *DeepSeekProvider) VideoDelete(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoDeleteRequest) (*schemas.BifrostVideoDeleteResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoDeleteRequest, provider.GetProviderKey()) +} + +// VideoList is not supported by DeepSeek provider. +func (provider *DeepSeekProvider) VideoList(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoListRequest) (*schemas.BifrostVideoListResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoListRequest, provider.GetProviderKey()) +} + +// VideoRemix is not supported by DeepSeek provider. +func (provider *DeepSeekProvider) VideoRemix(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostVideoRemixRequest) (*schemas.BifrostVideoGenerationResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.VideoRemixRequest, provider.GetProviderKey()) +} + +// FileUpload is not supported by DeepSeek provider. +func (provider *DeepSeekProvider) FileUpload(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostFileUploadRequest) (*schemas.BifrostFileUploadResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.FileUploadRequest, provider.GetProviderKey()) +} + +// FileList is not supported by DeepSeek provider. +func (provider *DeepSeekProvider) FileList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileListRequest) (*schemas.BifrostFileListResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.FileListRequest, provider.GetProviderKey()) +} + +// FileRetrieve is not supported by DeepSeek provider. +func (provider *DeepSeekProvider) FileRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileRetrieveRequest) (*schemas.BifrostFileRetrieveResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.FileRetrieveRequest, provider.GetProviderKey()) +} + +// FileDelete is not supported by DeepSeek provider. +func (provider *DeepSeekProvider) FileDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileDeleteRequest) (*schemas.BifrostFileDeleteResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.FileDeleteRequest, provider.GetProviderKey()) +} + +// FileContent is not supported by DeepSeek provider. +func (provider *DeepSeekProvider) FileContent(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostFileContentRequest) (*schemas.BifrostFileContentResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.FileContentRequest, provider.GetProviderKey()) +} + +// BatchCreate is not supported by DeepSeek provider. +func (provider *DeepSeekProvider) BatchCreate(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostBatchCreateRequest) (*schemas.BifrostBatchCreateResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchCreateRequest, provider.GetProviderKey()) +} + +// BatchList is not supported by DeepSeek provider. +func (provider *DeepSeekProvider) BatchList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchListRequest) (*schemas.BifrostBatchListResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchListRequest, provider.GetProviderKey()) +} + +// BatchRetrieve is not supported by DeepSeek provider. +func (provider *DeepSeekProvider) BatchRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchRetrieveRequest) (*schemas.BifrostBatchRetrieveResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchRetrieveRequest, provider.GetProviderKey()) +} + +// BatchCancel is not supported by DeepSeek provider. +func (provider *DeepSeekProvider) BatchCancel(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchCancelRequest) (*schemas.BifrostBatchCancelResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchCancelRequest, provider.GetProviderKey()) +} + +// BatchDelete is not supported by DeepSeek provider. +func (provider *DeepSeekProvider) BatchDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchDeleteRequest) (*schemas.BifrostBatchDeleteResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchDeleteRequest, provider.GetProviderKey()) +} + +// BatchResults is not supported by DeepSeek provider. +func (provider *DeepSeekProvider) BatchResults(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostBatchResultsRequest) (*schemas.BifrostBatchResultsResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.BatchResultsRequest, provider.GetProviderKey()) +} + +// CountTokens is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) CountTokens(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostResponsesRequest) (*schemas.BifrostCountTokensResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.CountTokensRequest, provider.GetProviderKey()) +} + +// Compaction is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) Compaction(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostCompactionRequest) (*schemas.BifrostCompactionResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.CompactionRequest, provider.GetProviderKey()) +} + +// ContainerCreate is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) ContainerCreate(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostContainerCreateRequest) (*schemas.BifrostContainerCreateResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerCreateRequest, provider.GetProviderKey()) +} + +// ContainerList is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) ContainerList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerListRequest) (*schemas.BifrostContainerListResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerListRequest, provider.GetProviderKey()) +} + +// ContainerRetrieve is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) ContainerRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerRetrieveRequest) (*schemas.BifrostContainerRetrieveResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerRetrieveRequest, provider.GetProviderKey()) +} + +// ContainerDelete is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) ContainerDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerDeleteRequest) (*schemas.BifrostContainerDeleteResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerDeleteRequest, provider.GetProviderKey()) +} + +// ContainerFileCreate is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) ContainerFileCreate(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostContainerFileCreateRequest) (*schemas.BifrostContainerFileCreateResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileCreateRequest, provider.GetProviderKey()) +} + +// ContainerFileList is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) ContainerFileList(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileListRequest) (*schemas.BifrostContainerFileListResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileListRequest, provider.GetProviderKey()) +} + +// ContainerFileRetrieve is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) ContainerFileRetrieve(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileRetrieveRequest) (*schemas.BifrostContainerFileRetrieveResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileRetrieveRequest, provider.GetProviderKey()) +} + +// ContainerFileContent is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) ContainerFileContent(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileContentRequest) (*schemas.BifrostContainerFileContentResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileContentRequest, provider.GetProviderKey()) +} + +// ContainerFileDelete is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) ContainerFileDelete(_ *schemas.BifrostContext, _ []schemas.Key, _ *schemas.BifrostContainerFileDeleteRequest) (*schemas.BifrostContainerFileDeleteResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.ContainerFileDeleteRequest, provider.GetProviderKey()) +} + +// Passthrough is not supported by the DeepSeek provider. +func (provider *DeepSeekProvider) Passthrough(_ *schemas.BifrostContext, _ schemas.Key, _ *schemas.BifrostPassthroughRequest) (*schemas.BifrostPassthroughResponse, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.PassthroughRequest, provider.GetProviderKey()) +} + +func (provider *DeepSeekProvider) PassthroughStream(_ *schemas.BifrostContext, _ schemas.PostHookRunner, _ func(context.Context), _ schemas.Key, _ *schemas.BifrostPassthroughRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + return nil, providerUtils.NewUnsupportedOperationError(schemas.PassthroughStreamRequest, provider.GetProviderKey()) +} diff --git a/core/providers/deepseek/deepseek_test.go b/core/providers/deepseek/deepseek_test.go new file mode 100644 index 00000000000..76d173cbf23 --- /dev/null +++ b/core/providers/deepseek/deepseek_test.go @@ -0,0 +1,60 @@ +package deepseek_test + +import ( + "os" + "strings" + "testing" + + "github.com/maximhq/bifrost/core/internal/llmtests" + + "github.com/maximhq/bifrost/core/schemas" +) + +func TestDeepseek(t *testing.T) { + t.Parallel() + if strings.TrimSpace(os.Getenv("DEEPSEEK_API_KEY")) == "" { + t.Skip("Skipping DeepSeek tests because DEEPSEEK_API_KEY is not set") + } + + client, ctx, cancel, err := llmtests.SetupTest() + if err != nil { + t.Fatalf("Error initializing test setup: %v", err) + } + defer cancel() + defer client.Shutdown() + + testConfig := llmtests.ComprehensiveTestConfig{ + Provider: schemas.DeepSeek, + ChatModel: "deepseek-v4-flash", + Fallbacks: []schemas.Fallback{ + {Provider: schemas.DeepSeek, Model: "deepseek-v4-flash"}, + {Provider: schemas.DeepSeek, Model: "deepseek-v4-pro"}, + }, + TextModel: "deepseek-v4-pro", + EmbeddingModel: "", // DeepSeek doesn't support embedding + ReasoningModel: "deepseek-v4-pro", + Scenarios: llmtests.TestScenarios{ + TextCompletion: true, + TextCompletionStream: true, + SimpleChat: true, + CompletionStream: true, + MultiTurnConversation: true, + ToolCalls: true, + ToolCallsStreaming: true, + MultipleToolCalls: false, + End2EndToolCalling: true, + AutomaticFunctionCall: true, + ImageURL: false, + ImageBase64: false, + MultipleImages: false, + CompleteEnd2End: true, + Embedding: false, + ListModels: true, + Reasoning: true, + }, + } + + t.Run("DeepSeekTests", func(t *testing.T) { + llmtests.RunAllComprehensiveTests(t, client, ctx, testConfig) + }) +} diff --git a/core/providers/elevenlabs/elevenlabs.go b/core/providers/elevenlabs/elevenlabs.go index 7140d7923c1..237e27050fe 100644 --- a/core/providers/elevenlabs/elevenlabs.go +++ b/core/providers/elevenlabs/elevenlabs.go @@ -103,7 +103,7 @@ func (provider *ElevenlabsProvider) listModelsByKey(ctx *schemas.BifrostContext, // Extract and set provider response headers so they're available on error paths ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerUtils.ExtractProviderResponseHeaders(resp)) if resp.StatusCode() != fasthttp.StatusOK { - return nil, parseElevenlabsError(resp) + return nil, providerUtils.SetErrorLatency(parseElevenlabsError(resp), latency) } var elevenlabsResponse ElevenlabsListModelsResponse @@ -237,20 +237,20 @@ func (provider *ElevenlabsProvider) Speech(ctx *schemas.BifrostContext, key sche latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Extract and set provider response headers so they're available on error paths ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerUtils.ExtractProviderResponseHeaders(resp)) // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, providerUtils.EnrichError(ctx, parseElevenlabsError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseElevenlabsError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Get the response body body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Create response based on whether timestamps were requested @@ -351,6 +351,7 @@ func (provider *ElevenlabsProvider) SpeechStream(ctx *schemas.BifrostContext, po // Make request startTime := time.Now() err := provider.streamingClient.Do(req, resp) + latency := time.Since(startTime) if err != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { @@ -361,16 +362,16 @@ func (provider *ElevenlabsProvider) SpeechStream(ctx *schemas.BifrostContext, po Message: schemas.ErrRequestCancelled, Error: err, }, - }, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + }, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Request failed before the first response byte (server closed an idle/pooled connection, // broken pipe, connection refused, DNS failure, etc.). Surface as a retriable upstream // connection error (502) so executeRequestWithRetries honors max_retries, matching the // non-streaming path - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Extract provider response headers before status check so error responses also forward them @@ -379,7 +380,7 @@ func (provider *ElevenlabsProvider) SpeechStream(ctx *schemas.BifrostContext, po // Check for HTTP errors if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) - return nil, providerUtils.EnrichError(ctx, parseElevenlabsError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseElevenlabsError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Create response channel @@ -543,7 +544,7 @@ func (provider *ElevenlabsProvider) Transcription(ctx *schemas.BifrostContext, k // Extract and set provider response headers so they're available on error paths ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerUtils.ExtractProviderResponseHeaders(resp)) if resp.StatusCode() != fasthttp.StatusOK { - return nil, parseElevenlabsError(resp) + return nil, providerUtils.SetErrorLatency(parseElevenlabsError(resp), latency) } responseBody, err := providerUtils.CheckAndDecodeBody(resp) diff --git a/core/providers/fireworks/fireworks.go b/core/providers/fireworks/fireworks.go index 9e8bd3ecd86..e4067c9de18 100644 --- a/core/providers/fireworks/fireworks.go +++ b/core/providers/fireworks/fireworks.go @@ -89,7 +89,6 @@ func (provider *FireworksProvider) listModelsByKey(_ *schemas.BifrostContext, ke ), nil } - // TextCompletion performs a text completion request to the Fireworks AI API. func (provider *FireworksProvider) TextCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostTextCompletionRequest) (*schemas.BifrostTextCompletionResponse, *schemas.BifrostError) { return openai.HandleOpenAITextCompletionRequest( @@ -97,7 +96,7 @@ func (provider *FireworksProvider) TextCompletion(ctx *schemas.BifrostContext, k provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -110,16 +109,12 @@ func (provider *FireworksProvider) TextCompletion(ctx *schemas.BifrostContext, k // TextCompletionStream performs a streaming text completion request to the Fireworks AI API. func (provider *FireworksProvider) TextCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostTextCompletionRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - if v := key.Value.GetValue(); v != "" { - authHeader = map[string]string{"Authorization": "Bearer " + v} - } return openai.HandleOpenAITextCompletionStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/completions"), request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -141,13 +136,14 @@ func (provider *FireworksProvider) ChatCompletion(ctx *schemas.BifrostContext, k provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), nil, nil, + nil, provider.logger, ) } @@ -157,17 +153,12 @@ func (provider *FireworksProvider) ChatCompletion(ctx *schemas.BifrostContext, k // Uses Fireworks AI's OpenAI-compatible streaming format. // Returns a channel containing BifrostStreamChunk objects representing the stream or an error if the request fails. func (provider *FireworksProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - if v := key.Value.GetValue(); v != "" { - authHeader = map[string]string{"Authorization": "Bearer " + v} - } - // Use shared OpenAI-compatible streaming logic return openai.HandleOpenAIChatCompletionStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -179,6 +170,7 @@ func (provider *FireworksProvider) ChatCompletionStream(ctx *schemas.BifrostCont nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -191,29 +183,26 @@ func (provider *FireworksProvider) Responses(ctx *schemas.BifrostContext, key sc provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/responses"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), nil, nil, + nil, provider.logger, ) } // ResponsesStream performs a streaming responses request to the Fireworks AI API. func (provider *FireworksProvider) ResponsesStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostResponsesRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - if v := key.Value.GetValue(); v != "" { - authHeader = map[string]string{"Authorization": "Bearer " + v} - } return openai.HandleOpenAIResponsesStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/responses"), request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -224,6 +213,7 @@ func (provider *FireworksProvider) ResponsesStream(ctx *schemas.BifrostContext, nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -236,7 +226,7 @@ func (provider *FireworksProvider) Embedding(ctx *schemas.BifrostContext, key sc provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/embeddings"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), diff --git a/core/providers/gemini/batch.go b/core/providers/gemini/batch.go index 387b0b58e53..8bf5dadf331 100644 --- a/core/providers/gemini/batch.go +++ b/core/providers/gemini/batch.go @@ -39,7 +39,10 @@ func ToGeminiBatchGenerateContentRequest(body map[string]interface{}) (GeminiBat if err := sonic.Unmarshal(messagesBytes, &chatMessages); err != nil { return geminiReq, fmt.Errorf("failed to unmarshal messages: %w", err) } - contents, systemInstruction := convertBifrostMessagesToGemini(chatMessages) + contents, systemInstruction, err := convertBifrostMessagesToGemini(chatMessages) + if err != nil { + return geminiReq, fmt.Errorf("failed to convert messages: %w", err) + } geminiReq.Contents = contents geminiReq.SystemInstruction = systemInstruction } @@ -301,7 +304,7 @@ func (provider *GeminiProvider) downloadBatchResultsFile(ctx context.Context, ke } // Make request - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { return nil, nil, bifrostErr @@ -309,7 +312,7 @@ func (provider *GeminiProvider) downloadBatchResultsFile(ctx context.Context, ke // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, nil, parseGeminiError(resp) + return nil, nil, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -343,34 +346,9 @@ func (provider *GeminiProvider) downloadBatchResultsFile(ctx context.Context, ke Message: resultLine.Error.Message, } } else if resultLine.Response != nil { - // Convert the response to a map for the Body field - respBody := make(map[string]interface{}) - if len(resultLine.Response.Candidates) > 0 { - candidate := resultLine.Response.Candidates[0] - if candidate.Content != nil && len(candidate.Content.Parts) > 0 { - var textParts []string - for _, part := range candidate.Content.Parts { - if part.Text != "" { - textParts = append(textParts, part.Text) - } - } - if len(textParts) > 0 { - respBody["text"] = strings.Join(textParts, "") - } - } - respBody["finish_reason"] = string(candidate.FinishReason) - } - if resultLine.Response.UsageMetadata != nil { - respBody["usage"] = map[string]interface{}{ - "prompt_tokens": resultLine.Response.UsageMetadata.PromptTokenCount, - "completion_tokens": resultLine.Response.UsageMetadata.CandidatesTokenCount, - "total_tokens": resultLine.Response.UsageMetadata.TotalTokenCount, - } - } - resultItem.Response = &schemas.BatchResultResponse{ StatusCode: 200, - Body: respBody, + Body: geminiGenerateContentToBatchResultBody(resultLine.Response), } } @@ -381,6 +359,86 @@ func (provider *GeminiProvider) downloadBatchResultsFile(ctx context.Context, ke return results, parseResult.Errors, nil } +// geminiBatchOutput extracts the batch output (a responses file name or inline responses) +// from a Gemini batch job response. The generativelanguage REST API reports the output +// under the Operation's top-level `response` field, mirrored in `metadata.output`; the +// `dest` field is a client-SDK-only abstraction and is never present on the wire. The +// top-level response is preferred, with metadata.output as a fallback. +func geminiBatchOutput(resp *GeminiBatchJobResponse) (fileName string, inlined []GeminiInlinedResponse) { + if resp == nil { + return "", nil + } + if resp.Response != nil { + fileName = resp.Response.ResponsesFile + if resp.Response.InlinedResponses != nil { + inlined = resp.Response.InlinedResponses.InlinedResponses + } + } + if fileName == "" && len(inlined) == 0 && resp.Metadata != nil && resp.Metadata.Output != nil { + fileName = resp.Metadata.Output.ResponsesFile + if resp.Metadata.Output.InlinedResponses != nil { + inlined = resp.Metadata.Output.InlinedResponses.InlinedResponses + } + } + return fileName, inlined +} + +// geminiGenerateContentToBatchResultBody flattens a Gemini GenerateContentResponse into +// the compact result body shape shared by the inline and file-based batch result paths. +func geminiGenerateContentToBatchResultBody(resp *GenerateContentResponse) map[string]interface{} { + body := make(map[string]interface{}) + if resp == nil { + return body + } + if len(resp.Candidates) > 0 { + candidate := resp.Candidates[0] + if candidate.Content != nil && len(candidate.Content.Parts) > 0 { + var textParts []string + for _, part := range candidate.Content.Parts { + if part.Text != "" { + textParts = append(textParts, part.Text) + } + } + if len(textParts) > 0 { + body["text"] = strings.Join(textParts, "") + } + } + body["finish_reason"] = string(candidate.FinishReason) + } + if resp.UsageMetadata != nil { + body["usage"] = map[string]interface{}{ + "prompt_tokens": resp.UsageMetadata.PromptTokenCount, + "completion_tokens": resp.UsageMetadata.CandidatesTokenCount, + "total_tokens": resp.UsageMetadata.TotalTokenCount, + } + } + return body +} + +// geminiInlineResponseToBatchResultItem converts a single Gemini inline batch response +// into a Bifrost BatchResultItem. customIDFallback is used when the response carries no +// metadata key. +func geminiInlineResponseToBatchResultItem(inlineResp GeminiInlinedResponse, customIDFallback string) schemas.BatchResultItem { + customID := customIDFallback + if inlineResp.Metadata != nil && inlineResp.Metadata.Key != "" { + customID = inlineResp.Metadata.Key + } + + resultItem := schemas.BatchResultItem{CustomID: customID} + if inlineResp.Error != nil { + resultItem.Error = &schemas.BatchResultError{ + Code: fmt.Sprintf("%d", inlineResp.Error.Code), + Message: inlineResp.Error.Message, + } + } else if inlineResp.Response != nil { + resultItem.Response = &schemas.BatchResultResponse{ + StatusCode: 200, + Body: geminiGenerateContentToBatchResultBody(inlineResp.Response), + } + } + return resultItem +} + // extractGeminiUsageMetadata extracts usage metadata (as ints) from Gemini response func extractGeminiUsageMetadata(geminiResponse *GenerateContentResponse) (int, int, int) { var inputTokens, outputTokens, totalTokens int diff --git a/core/providers/gemini/batchresults_test.go b/core/providers/gemini/batchresults_test.go new file mode 100644 index 00000000000..81c5eedabb2 --- /dev/null +++ b/core/providers/gemini/batchresults_test.go @@ -0,0 +1,195 @@ +package gemini + +import ( + "testing" + + "github.com/bytedance/sonic" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// TestGeminiBatchOutput locks in the fix for issue #3951: the generativelanguage batch +// REST API reports output under the Operation's `response` field (mirrored in +// `metadata.output`), never under `dest`. Inline responses are nested one level deep +// (response.inlinedResponses.inlinedResponses). geminiBatchOutput must read the real +// fields so completed results are no longer silently dropped. +func TestGeminiBatchOutput(t *testing.T) { + t.Run("InlineUnderResponse", func(t *testing.T) { + raw := `{ + "name": "batches/abc123", + "metadata": { + "@type": "type.googleapis.com/google.ai.generativelanguage.v1beta.GenerateContentBatch", + "name": "batches/abc123", + "state": "BATCH_STATE_SUCCEEDED", + "batchStats": {"requestCount": "2", "successfulRequestCount": "2", "pendingRequestCount": "0"} + }, + "done": true, + "response": { + "@type": "type.googleapis.com/google.ai.generativelanguage.v1beta.GenerateContentBatchOutput", + "inlinedResponses": { + "inlinedResponses": [ + {"metadata": {"key": "req-1"}, "response": {"candidates": [{"content": {"parts": [{"text": "4"}]}, "finishReason": "STOP"}], "usageMetadata": {"promptTokenCount": 5, "candidatesTokenCount": 1, "totalTokenCount": 6}}}, + {"metadata": {"key": "req-2"}, "response": {"candidates": [{"content": {"parts": [{"text": "Paris"}]}, "finishReason": "STOP"}], "usageMetadata": {"promptTokenCount": 4, "candidatesTokenCount": 1, "totalTokenCount": 5}}} + ] + } + } +}` + var resp GeminiBatchJobResponse + require.NoError(t, sonic.Unmarshal([]byte(raw), &resp)) + + fileName, inlined := geminiBatchOutput(&resp) + assert.Empty(t, fileName) + require.Len(t, inlined, 2) + + require.NotNil(t, inlined[0].Metadata) + assert.Equal(t, "req-1", inlined[0].Metadata.Key) + require.NotNil(t, inlined[0].Response) + require.Len(t, inlined[0].Response.Candidates, 1) + require.NotNil(t, inlined[0].Response.Candidates[0].Content) + require.Len(t, inlined[0].Response.Candidates[0].Content.Parts, 1) + assert.Equal(t, "4", inlined[0].Response.Candidates[0].Content.Parts[0].Text) + + require.NotNil(t, inlined[1].Metadata) + assert.Equal(t, "req-2", inlined[1].Metadata.Key) + }) + + t.Run("FileUnderResponse", func(t *testing.T) { + raw := `{ + "name": "batches/file1", + "metadata": {"name": "batches/file1", "state": "BATCH_STATE_SUCCEEDED", "batchStats": {"requestCount": "10", "successfulRequestCount": "10", "pendingRequestCount": "0"}}, + "done": true, + "response": {"@type": "type.googleapis.com/google.ai.generativelanguage.v1beta.GenerateContentBatchOutput", "responsesFile": "files/batch-out-1"} +}` + var resp GeminiBatchJobResponse + require.NoError(t, sonic.Unmarshal([]byte(raw), &resp)) + + fileName, inlined := geminiBatchOutput(&resp) + assert.Equal(t, "files/batch-out-1", fileName) + assert.Empty(t, inlined) + }) + + t.Run("MetadataOutputFallback", func(t *testing.T) { + // No top-level response; output only present under metadata.output. + raw := `{ + "name": "batches/meta1", + "metadata": { + "name": "batches/meta1", + "state": "BATCH_STATE_SUCCEEDED", + "batchStats": {"requestCount": "1", "successfulRequestCount": "1", "pendingRequestCount": "0"}, + "output": {"inlinedResponses": {"inlinedResponses": [{"metadata": {"key": "only-1"}, "response": {"candidates": [{"content": {"parts": [{"text": "hey"}]}, "finishReason": "STOP"}]}}]}} + }, + "done": true +}` + var resp GeminiBatchJobResponse + require.NoError(t, sonic.Unmarshal([]byte(raw), &resp)) + + fileName, inlined := geminiBatchOutput(&resp) + assert.Empty(t, fileName) + require.Len(t, inlined, 1) + require.NotNil(t, inlined[0].Metadata) + assert.Equal(t, "only-1", inlined[0].Metadata.Key) + }) + + t.Run("IgnoresDestField", func(t *testing.T) { + // The legacy dest field must never be read: only response/metadata.output count. + resp := &GeminiBatchJobResponse{ + Name: "batches/dest1", + Dest: &GeminiBatchDest{FileName: "files/should-be-ignored"}, + } + fileName, inlined := geminiBatchOutput(resp) + assert.Empty(t, fileName) + assert.Empty(t, inlined) + }) + + t.Run("NilResponse", func(t *testing.T) { + fileName, inlined := geminiBatchOutput(nil) + assert.Empty(t, fileName) + assert.Empty(t, inlined) + }) +} + +// TestGeminiInlineResponseToBatchResultItem verifies the per-response conversion used by +// the batch results path for inline batches. +func TestGeminiInlineResponseToBatchResultItem(t *testing.T) { + t.Run("SuccessWithMetadataKey", func(t *testing.T) { + inline := GeminiInlinedResponse{ + Metadata: &GeminiBatchMetadata{Key: "row-9"}, + Response: &GenerateContentResponse{ + Candidates: []*Candidate{{ + Content: &Content{Parts: []*Part{{Text: "hi"}}}, + FinishReason: FinishReasonStop, + }}, + UsageMetadata: &GenerateContentResponseUsageMetadata{ + PromptTokenCount: 2, + CandidatesTokenCount: 3, + TotalTokenCount: 5, + }, + }, + } + + item := geminiInlineResponseToBatchResultItem(inline, "request-0") + assert.Equal(t, "row-9", item.CustomID) + assert.Nil(t, item.Error) + require.NotNil(t, item.Response) + assert.Equal(t, 200, item.Response.StatusCode) + assert.Equal(t, "hi", item.Response.Body["text"]) + assert.Equal(t, "STOP", item.Response.Body["finish_reason"]) + + usage, ok := item.Response.Body["usage"].(map[string]interface{}) + require.True(t, ok) + assert.EqualValues(t, 2, usage["prompt_tokens"]) + assert.EqualValues(t, 3, usage["completion_tokens"]) + assert.EqualValues(t, 5, usage["total_tokens"]) + }) + + t.Run("ErrorUsesFallbackID", func(t *testing.T) { + inline := GeminiInlinedResponse{ + Error: &GeminiBatchErrorInfo{Code: 429, Message: "rate limited"}, + } + + item := geminiInlineResponseToBatchResultItem(inline, "request-3") + assert.Equal(t, "request-3", item.CustomID) + assert.Nil(t, item.Response) + require.NotNil(t, item.Error) + assert.Equal(t, "429", item.Error.Code) + assert.Equal(t, "rate limited", item.Error.Message) + }) +} + +// TestGeminiGenerateContentToBatchResultBody verifies the flattening of a Gemini +// GenerateContentResponse into the compact batch result body shared by inline and +// file-based result paths. +func TestGeminiGenerateContentToBatchResultBody(t *testing.T) { + t.Run("TextAndUsage", func(t *testing.T) { + resp := &GenerateContentResponse{ + Candidates: []*Candidate{{ + Content: &Content{Parts: []*Part{{Text: "foo"}, {Text: "bar"}}}, + FinishReason: FinishReasonStop, + }}, + UsageMetadata: &GenerateContentResponseUsageMetadata{ + PromptTokenCount: 1, + CandidatesTokenCount: 2, + TotalTokenCount: 3, + }, + } + + body := geminiGenerateContentToBatchResultBody(resp) + assert.Equal(t, "foobar", body["text"]) + assert.Equal(t, "STOP", body["finish_reason"]) + _, ok := body["usage"].(map[string]interface{}) + assert.True(t, ok) + }) + + t.Run("NoUsageNoText", func(t *testing.T) { + resp := &GenerateContentResponse{ + Candidates: []*Candidate{{FinishReason: FinishReasonMaxTokens}}, + } + + body := geminiGenerateContentToBatchResultBody(resp) + assert.Equal(t, "MAX_TOKENS", body["finish_reason"]) + _, hasText := body["text"] + assert.False(t, hasText) + _, hasUsage := body["usage"] + assert.False(t, hasUsage) + }) +} diff --git a/core/providers/gemini/cachedcontents.go b/core/providers/gemini/cachedcontents.go index 74375f90fd2..3c5662b2ba4 100644 --- a/core/providers/gemini/cachedcontents.go +++ b/core/providers/gemini/cachedcontents.go @@ -249,7 +249,7 @@ func (provider *GeminiProvider) CachedContentCreate(ctx *schemas.BifrostContext, return nil, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK { - return nil, parseGeminiError(resp) + return nil, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } respBody, decErr := providerUtils.CheckAndDecodeBody(resp) @@ -313,7 +313,7 @@ func (provider *GeminiProvider) cachedContentListByKey(ctx *schemas.BifrostConte return nil, latency, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK { - return nil, latency, parseGeminiError(resp) + return nil, latency, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } respBody, decErr := providerUtils.CheckAndDecodeBody(resp) @@ -382,7 +382,7 @@ func (provider *GeminiProvider) cachedContentRetrieveByKey(ctx *schemas.BifrostC return nil, latency, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK { - return nil, latency, parseGeminiError(resp) + return nil, latency, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } respBody, decErr := providerUtils.CheckAndDecodeBody(resp) @@ -482,7 +482,7 @@ func (provider *GeminiProvider) cachedContentUpdateByKey(ctx *schemas.BifrostCon return nil, latency, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK { - return nil, latency, parseGeminiError(resp) + return nil, latency, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } respBody, decErr := providerUtils.CheckAndDecodeBody(resp) @@ -563,7 +563,7 @@ func (provider *GeminiProvider) cachedContentDeleteByKey(ctx *schemas.BifrostCon return nil, latency, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK { - return nil, latency, parseGeminiError(resp) + return nil, latency, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } return &schemas.BifrostCachedContentDeleteResponse{ diff --git a/core/providers/gemini/chat.go b/core/providers/gemini/chat.go index b56b180743a..1555afca6de 100644 --- a/core/providers/gemini/chat.go +++ b/core/providers/gemini/chat.go @@ -10,6 +10,12 @@ import ( // ToGeminiChatCompletionRequest converts a BifrostChatRequest to Gemini's generation request format for chat completion func ToGeminiChatCompletionRequest(ctx *schemas.BifrostContext, bifrostReq *schemas.BifrostChatRequest) (*GeminiGenerationRequest, error) { + return ToGeminiChatCompletionRequestWithImageURLSchemes(ctx, bifrostReq, defaultGeminiImageURLSchemes...) +} + +// ToGeminiChatCompletionRequestWithImageURLSchemes converts a BifrostChatRequest +// to Gemini format using the provider-specific allowlist for non-data image URLs. +func ToGeminiChatCompletionRequestWithImageURLSchemes(ctx *schemas.BifrostContext, bifrostReq *schemas.BifrostChatRequest, allowedImageURLSchemes ...string) (*GeminiGenerationRequest, error) { if bifrostReq == nil { return nil, nil } @@ -75,7 +81,10 @@ func ToGeminiChatCompletionRequest(ctx *schemas.BifrostContext, bifrostReq *sche } } // Convert chat completion messages to Gemini format - contents, systemInstruction := convertBifrostMessagesToGemini(bifrostReq.Input) + contents, systemInstruction, err := convertBifrostMessagesToGemini(bifrostReq.Input, allowedImageURLSchemes...) + if err != nil { + return nil, err + } if systemInstruction != nil { geminiReq.SystemInstruction = systemInstruction } diff --git a/core/providers/gemini/errors.go b/core/providers/gemini/errors.go index 2d60a7bcd36..205d6b6390b 100644 --- a/core/providers/gemini/errors.go +++ b/core/providers/gemini/errors.go @@ -59,6 +59,9 @@ func parseGeminiError(resp *fasthttp.Response) *schemas.BifrostError { // Set Code from first error if available if firstError != nil { bifrostErr.Error.Code = schemas.Ptr(strconv.Itoa(firstError.Code)) + if firstError.Status != "" { + bifrostErr.Error.Type = schemas.Ptr(firstError.Status) + } } // Set Message to trimmed concatenated message bifrostErr.Error.Message = message @@ -74,6 +77,9 @@ func parseGeminiError(resp *fasthttp.Response) *schemas.BifrostError { } bifrostErr.Error.Code = schemas.Ptr(strconv.Itoa(errorResp.Error.Code)) bifrostErr.Error.Message = errorResp.Error.Message + if errorResp.Error.Status != "" { + bifrostErr.Error.Type = schemas.Ptr(errorResp.Error.Status) + } } return bifrostErr } diff --git a/core/providers/gemini/errors_test.go b/core/providers/gemini/errors_test.go new file mode 100644 index 00000000000..dab7c954a9d --- /dev/null +++ b/core/providers/gemini/errors_test.go @@ -0,0 +1,62 @@ +package gemini + +import ( + "testing" + + "github.com/valyala/fasthttp" +) + +// TestParseGeminiError_SingleObjectPopulatesStatusType verifies the Gemini +// status (e.g. RESOURCE_EXHAUSTED) is surfaced on error.type when the body is a +// single {"error":{...}} object. +func TestParseGeminiError_SingleObjectPopulatesStatusType(t *testing.T) { + var resp fasthttp.Response + resp.SetStatusCode(fasthttp.StatusTooManyRequests) + resp.SetBodyString(`{"error":{"code":429,"message":"Quota exceeded","status":"RESOURCE_EXHAUSTED"}}`) + + bifrostErr := parseGeminiError(&resp) + + if bifrostErr == nil || bifrostErr.Error == nil { + t.Fatal("expected non-nil error response") + } + if bifrostErr.Error.Type == nil || *bifrostErr.Error.Type != "RESOURCE_EXHAUSTED" { + t.Fatalf("expected error.type RESOURCE_EXHAUSTED, got %v", bifrostErr.Error.Type) + } +} + +// TestParseGeminiError_ArrayPopulatesStatusType verifies the status is surfaced +// on error.type when the body is an array of errors. +func TestParseGeminiError_ArrayPopulatesStatusType(t *testing.T) { + var resp fasthttp.Response + resp.SetStatusCode(fasthttp.StatusBadRequest) + resp.SetBodyString(`[{"error":{"code":400,"message":"bad request","status":"INVALID_ARGUMENT"}}]`) + + bifrostErr := parseGeminiError(&resp) + + if bifrostErr == nil || bifrostErr.Error == nil { + t.Fatal("expected non-nil error response") + } + if bifrostErr.Error.Type == nil || *bifrostErr.Error.Type != "INVALID_ARGUMENT" { + t.Fatalf("expected error.type INVALID_ARGUMENT, got %v", bifrostErr.Error.Type) + } +} + +// TestParseGeminiError_RoundTripToGeminiError is the regression test for the +// broken passthrough: ToGeminiError reconstructs the status field from +// error.type, so the round trip must preserve the Gemini status rather than +// returning an empty status. +func TestParseGeminiError_RoundTripToGeminiError(t *testing.T) { + var resp fasthttp.Response + resp.SetStatusCode(fasthttp.StatusTooManyRequests) + resp.SetBodyString(`{"error":{"code":429,"message":"Quota exceeded","status":"RESOURCE_EXHAUSTED"}}`) + + bifrostErr := parseGeminiError(&resp) + geminiErr := ToGeminiError(bifrostErr) + + if geminiErr == nil || geminiErr.Error == nil { + t.Fatal("expected non-nil gemini error") + } + if geminiErr.Error.Status != "RESOURCE_EXHAUSTED" { + t.Fatalf("expected status RESOURCE_EXHAUSTED to survive round trip, got %q", geminiErr.Error.Status) + } +} diff --git a/core/providers/gemini/fileupload_test.go b/core/providers/gemini/fileupload_test.go new file mode 100644 index 00000000000..1e48f3c74a1 --- /dev/null +++ b/core/providers/gemini/fileupload_test.go @@ -0,0 +1,123 @@ +package gemini + +import ( + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "net/http/httptest" + "testing" + + "github.com/maximhq/bifrost/core/schemas" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func TestFileUploadSendsContentTypeToGemini(t *testing.T) { + const contentType = "application/pdf" + + ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodPost { + http.Error(w, "unexpected method", http.StatusMethodNotAllowed) + return + } + if r.URL.Path != "/upload/v1beta/files" { + http.Error(w, fmt.Sprintf("unexpected path: %s", r.URL.Path), http.StatusNotFound) + return + } + if got := r.Header.Get("x-goog-api-key"); got != "dummy-key" { + http.Error(w, fmt.Sprintf("unexpected api key: %s", got), http.StatusUnauthorized) + return + } + + reader, err := r.MultipartReader() + if err != nil { + http.Error(w, fmt.Sprintf("failed to read multipart body: %v", err), http.StatusBadRequest) + return + } + + var sawMetadata, sawFile bool + for { + part, err := reader.NextPart() + if err == io.EOF { + break + } + if err != nil { + http.Error(w, fmt.Sprintf("failed to read multipart part: %v", err), http.StatusBadRequest) + return + } + body, err := io.ReadAll(part) + if err != nil { + http.Error(w, fmt.Sprintf("failed to read multipart part body: %v", err), http.StatusBadRequest) + return + } + + switch part.FormName() { + case "metadata": + sawMetadata = true + var metadata struct { + File struct { + DisplayName string `json:"displayName"` + MIMEType string `json:"mimeType"` + } `json:"file"` + } + if err := json.Unmarshal(body, &metadata); err != nil { + http.Error(w, fmt.Sprintf("failed to unmarshal metadata: %v", err), http.StatusBadRequest) + return + } + if metadata.File.DisplayName != "tiny.pdf" { + http.Error(w, fmt.Sprintf("unexpected displayName: %s", metadata.File.DisplayName), http.StatusBadRequest) + return + } + if metadata.File.MIMEType != contentType { + http.Error(w, fmt.Sprintf("unexpected metadata mimeType: %s", metadata.File.MIMEType), http.StatusBadRequest) + return + } + case "file": + sawFile = true + if part.FileName() != "tiny.pdf" { + http.Error(w, fmt.Sprintf("unexpected filename: %s", part.FileName()), http.StatusBadRequest) + return + } + if got := part.Header.Get("Content-Type"); got != contentType { + http.Error(w, fmt.Sprintf("unexpected file part content type: %s", got), http.StatusBadRequest) + return + } + if string(body) != "ok" { + http.Error(w, fmt.Sprintf("unexpected file body: %s", string(body)), http.StatusBadRequest) + return + } + } + } + if !sawMetadata || !sawFile { + http.Error(w, "missing metadata or file multipart part", http.StatusBadRequest) + return + } + + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusOK) + _, _ = w.Write([]byte(`{"file":{"name":"files/test","displayName":"tiny.pdf","mimeType":"application/pdf","sizeBytes":"2","createTime":"2026-07-01T00:00:00Z","state":"ACTIVE","uri":"https://generativelanguage.googleapis.com/v1beta/files/test"}}`)) + })) + defer ts.Close() + + provider := NewGeminiProvider(&schemas.ProviderConfig{ + NetworkConfig: schemas.NetworkConfig{BaseURL: ts.URL + "/v1beta"}, + }, testNoopLogger{}) + + ctx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline) + key := schemas.Key{Value: *schemas.NewSecretVar("dummy-key")} + resp, bifrostErr := provider.FileUpload(ctx, key, &schemas.BifrostFileUploadRequest{ + Provider: schemas.Gemini, + File: []byte("ok"), + Filename: "tiny.pdf", + Purpose: schemas.FilePurposeUserData, + ContentType: schemas.Ptr(contentType), + }) + + require.Nil(t, bifrostErr) + require.NotNil(t, resp) + assert.Equal(t, "files/test", resp.ID) + assert.Equal(t, "tiny.pdf", resp.Filename) + assert.Equal(t, "https://generativelanguage.googleapis.com/v1beta/files/test", resp.StorageURI) +} diff --git a/core/providers/gemini/gemini.go b/core/providers/gemini/gemini.go index 278c4ac4fb4..38c8e91cac9 100644 --- a/core/providers/gemini/gemini.go +++ b/core/providers/gemini/gemini.go @@ -10,6 +10,7 @@ import ( "io" "mime/multipart" "net/http" + "net/textproto" "net/url" "strings" "time" @@ -152,7 +153,7 @@ func (provider *GeminiProvider) completeRequest(ctx *schemas.BifrostContext, mod // Handle error response if resp.StatusCode() != fasthttp.StatusOK { providerUtils.MaterializeStreamErrorBody(ctx, resp) - return nil, nil, latency, providerResponseHeaders, parseGeminiError(resp) + return nil, nil, latency, providerResponseHeaders, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } body, isLargeResp, decodeErr := providerUtils.FinalizeResponseWithLargeDetection(ctx, resp, provider.logger) @@ -213,7 +214,7 @@ func (provider *GeminiProvider) listModelsByKey(ctx *schemas.BifrostContext, key // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, parseGeminiError(resp) + return nil, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } // Parse Gemini's response @@ -299,7 +300,7 @@ func (provider *GeminiProvider) ChatCompletion(ctx *schemas.BifrostContext, key ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Large response mode: return lightweight response with metadata only @@ -366,8 +367,6 @@ func (provider *GeminiProvider) ChatCompletionStream(ctx *schemas.BifrostContext headers["x-goog-api-key"] = key.Value.GetValue() } - providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) - // Use shared Gemini streaming logic return HandleGeminiChatCompletionStream( ctx, @@ -376,6 +375,7 @@ func (provider *GeminiProvider) ChatCompletionStream(ctx *schemas.BifrostContext jsonData, headers, provider.networkConfig.ExtraHeaders, + provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), @@ -395,6 +395,7 @@ func HandleGeminiChatCompletionStream( jsonBody []byte, headers map[string]string, extraHeaders map[string]string, + streamIdleTimeoutInSeconds int, sendBackRawRequest bool, sendBackRawResponse bool, providerName schemas.ModelProvider, @@ -404,6 +405,7 @@ func HandleGeminiChatCompletionStream( logger schemas.Logger, postHookSpanFinalizer func(context.Context), ) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, streamIdleTimeoutInSeconds) req := fasthttp.AcquireRequest() resp := fasthttp.AcquireResponse() resp.StreamBody = true @@ -428,6 +430,7 @@ func HandleGeminiChatCompletionStream( startTime := time.Now() // Make the request — caller is responsible for passing a streaming-configured client. doErr := client.Do(req, resp) + latency := time.Since(startTime) if doErr != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(doErr, context.Canceled) { @@ -438,12 +441,12 @@ func HandleGeminiChatCompletionStream( Message: schemas.ErrRequestCancelled, Error: doErr, }, - }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } if errors.Is(doErr, fasthttp.ErrTimeout) || errors.Is(doErr, context.DeadlineExceeded) { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, doErr), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, doErr), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, doErr), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, doErr), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Extract provider response headers before status check so error responses also forward them @@ -453,7 +456,7 @@ func HandleGeminiChatCompletionStream( if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) respBody := append([]byte(nil), resp.Body()...) - return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonBody, respBody, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonBody, respBody, sendBackRawRequest, sendBackRawResponse, latency) } // Large payload streaming passthrough — pipe raw upstream SSE to client @@ -485,7 +488,7 @@ func HandleGeminiChatCompletionStream( fmt.Errorf("provider returned an empty response"), ) ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } @@ -566,7 +569,7 @@ func HandleGeminiChatCompletionStream( }, } ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } logger.Warn("Failed to process chunk: %v", err) @@ -585,7 +588,7 @@ func HandleGeminiChatCompletionStream( response, bifrostErr, isLastChunk := geminiResponse.ToBifrostChatCompletionStream(streamState) if bifrostErr != nil { ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } @@ -692,7 +695,7 @@ func (provider *GeminiProvider) Responses(ctx *schemas.BifrostContext, key schem ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Large response mode: return lightweight response with metadata only @@ -778,7 +781,7 @@ func (provider *GeminiProvider) responsesWithLargeResponseDetection( bifrostErr := parseGeminiError(resp) wait() fasthttp.ReleaseResponse(resp) - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Delegate large response detection + normal buffered path to shared utility @@ -786,7 +789,7 @@ func (provider *GeminiProvider) responsesWithLargeResponseDetection( if respErr != nil { wait() fasthttp.ReleaseResponse(resp) - return nil, providerUtils.EnrichError(ctx, respErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, respErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if isLarge { // Build lightweight response with usage from preview for plugin pipeline @@ -877,8 +880,6 @@ func (provider *GeminiProvider) ResponsesStream(ctx *schemas.BifrostContext, pos headers["x-goog-api-key"] = key.Value.GetValue() } - providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) - return HandleGeminiResponsesStream( ctx, provider.streamingClient, @@ -886,6 +887,7 @@ func (provider *GeminiProvider) ResponsesStream(ctx *schemas.BifrostContext, pos jsonData, headers, provider.networkConfig.ExtraHeaders, + provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), @@ -905,6 +907,7 @@ func HandleGeminiResponsesStream( jsonBody []byte, headers map[string]string, extraHeaders map[string]string, + streamIdleTimeoutInSeconds int, sendBackRawRequest bool, sendBackRawResponse bool, providerName schemas.ModelProvider, @@ -914,6 +917,7 @@ func HandleGeminiResponsesStream( logger schemas.Logger, postHookSpanFinalizer func(context.Context), ) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { + providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, streamIdleTimeoutInSeconds) req := fasthttp.AcquireRequest() resp := fasthttp.AcquireResponse() resp.StreamBody = true @@ -938,6 +942,7 @@ func HandleGeminiResponsesStream( startTime := time.Now() // Make the request — caller is responsible for passing a streaming-configured client. doErr := client.Do(req, resp) + latency := time.Since(startTime) if doErr != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(doErr, context.Canceled) { @@ -948,12 +953,12 @@ func HandleGeminiResponsesStream( Message: schemas.ErrRequestCancelled, Error: doErr, }, - }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } if errors.Is(doErr, fasthttp.ErrTimeout) || errors.Is(doErr, context.DeadlineExceeded) { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, doErr), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, doErr), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, doErr), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, doErr), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Extract provider response headers before status check so error responses also forward them @@ -962,7 +967,7 @@ func HandleGeminiResponsesStream( // Check for HTTP errors — use parseGeminiError to preserve upstream error details if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) - return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Large payload streaming passthrough — pipe raw upstream SSE to client @@ -1246,7 +1251,7 @@ func (provider *GeminiProvider) Embedding(ctx *schemas.BifrostContext, key schem if bifrostErr != nil { wait() fasthttp.ReleaseResponse(resp) - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // When upstream responds before consuming full upload, drain remaining bytes from // ingress reader so proxy hops (e.g., Caddy) don't surface broken-pipe 502s. @@ -1262,7 +1267,7 @@ func (provider *GeminiProvider) Embedding(ctx *schemas.BifrostContext, key schem if resp.StatusCode() != fasthttp.StatusOK { providerUtils.MaterializeStreamErrorBody(ctx, resp) provider.logger.Debug(fmt.Sprintf("error from %s provider: %s", providerName, string(resp.Body()))) - parsedErr := providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + parsedErr := providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) wait() fasthttp.ReleaseResponse(resp) return nil, parsedErr @@ -1272,7 +1277,7 @@ func (provider *GeminiProvider) Embedding(ctx *schemas.BifrostContext, key schem if decodeErr != nil { wait() fasthttp.ReleaseResponse(resp) - return nil, providerUtils.EnrichError(ctx, decodeErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, decodeErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if isLargeResp { // Large response detected — return lightweight response with metadata only; @@ -1295,7 +1300,7 @@ func (provider *GeminiProvider) Embedding(ctx *schemas.BifrostContext, key schem providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Convert to Bifrost format @@ -1344,7 +1349,7 @@ func (provider *GeminiProvider) Speech(ctx *schemas.BifrostContext, key schemas. ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Large response mode: return lightweight response with metadata only @@ -1362,7 +1367,7 @@ func (provider *GeminiProvider) Speech(ctx *schemas.BifrostContext, key schemas. } response, convErr := geminiResponse.ToBifrostSpeechResponse(ctx) if convErr != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, convErr), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, convErr), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Set ExtraFields @@ -1436,6 +1441,7 @@ func (provider *GeminiProvider) SpeechStream(ctx *schemas.BifrostContext, postHo startTime := time.Now() // Make the request err := provider.streamingClient.Do(req, resp) + latency := time.Since(startTime) if err != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { @@ -1446,16 +1452,16 @@ func (provider *GeminiProvider) SpeechStream(ctx *schemas.BifrostContext, postHo Message: schemas.ErrRequestCancelled, Error: err, }, - }, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + }, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Request failed before the first response byte (server closed an idle/pooled connection, // broken pipe, connection refused, DNS failure, etc.). Surface as a retriable upstream // connection error (502) so executeRequestWithRetries honors max_retries, matching the // non-streaming path - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Extract provider response headers before status check so error responses also forward them @@ -1464,9 +1470,11 @@ func (provider *GeminiProvider) SpeechStream(ctx *schemas.BifrostContext, postHo // Check for HTTP errors if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) - return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } + providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) + // Large payload streaming passthrough — pipe raw upstream SSE to client if providerUtils.SetupStreamingPassthrough(ctx, resp) { responseChan := make(chan *schemas.BifrostStreamChunk) @@ -1477,8 +1485,6 @@ func (provider *GeminiProvider) SpeechStream(ctx *schemas.BifrostContext, postHo // Create response channel responseChan := make(chan *schemas.BifrostStreamChunk, schemas.DefaultStreamBufferSize) - providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) - // Start streaming in a goroutine go func() { defer providerUtils.EnsureStreamFinalizerCalled(ctx, postHookSpanFinalizer) @@ -1654,7 +1660,7 @@ func (provider *GeminiProvider) Transcription(ctx *schemas.BifrostContext, key s ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Large response mode: return lightweight response with metadata only @@ -1730,6 +1736,7 @@ func (provider *GeminiProvider) TranscriptionStream(ctx *schemas.BifrostContext, startTime := time.Now() // Make the request err := provider.streamingClient.Do(req, resp) + latency := time.Since(startTime) if err != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { @@ -1740,16 +1747,16 @@ func (provider *GeminiProvider) TranscriptionStream(ctx *schemas.BifrostContext, Message: schemas.ErrRequestCancelled, Error: err, }, - }, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + }, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Request failed before the first response byte (server closed an idle/pooled connection, // broken pipe, connection refused, DNS failure, etc.). Surface as a retriable upstream // connection error (502) so executeRequestWithRetries honors max_retries, matching the // non-streaming path - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Extract provider response headers before status check so error responses also forward them @@ -1758,9 +1765,11 @@ func (provider *GeminiProvider) TranscriptionStream(ctx *schemas.BifrostContext, // Check for HTTP errors if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) - return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } + providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) + // Large payload streaming passthrough — pipe raw upstream SSE to client if providerUtils.SetupStreamingPassthrough(ctx, resp) { responseChan := make(chan *schemas.BifrostStreamChunk) @@ -1771,8 +1780,6 @@ func (provider *GeminiProvider) TranscriptionStream(ctx *schemas.BifrostContext, // Create response channel responseChan := make(chan *schemas.BifrostStreamChunk, schemas.DefaultStreamBufferSize) - providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) - // Start streaming in a goroutine go func() { defer providerUtils.EnsureStreamFinalizerCalled(ctx, postHookSpanFinalizer) @@ -1952,7 +1959,7 @@ func (provider *GeminiProvider) ImageGeneration(ctx *schemas.BifrostContext, key ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Large response mode: return lightweight response with metadata only @@ -2039,7 +2046,7 @@ func (provider *GeminiProvider) handleImagenImageGeneration(ctx *schemas.Bifrost // Handle error response if resp.StatusCode() != fasthttp.StatusOK { providerUtils.MaterializeStreamErrorBody(ctx, resp) - return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Parse Imagen response @@ -2129,7 +2136,7 @@ func (provider *GeminiProvider) ImageEdit(ctx *schemas.BifrostContext, key schem if resp.StatusCode() != fasthttp.StatusOK { providerUtils.MaterializeStreamErrorBody(ctx, resp) - return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } body, isLargeResp, decodeErr := providerUtils.FinalizeResponseWithLargeDetection(ctx, resp, provider.logger) @@ -2181,7 +2188,7 @@ func (provider *GeminiProvider) ImageEdit(ctx *schemas.BifrostContext, key schem ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Large response mode: return lightweight response with metadata only @@ -2279,13 +2286,13 @@ func (provider *GeminiProvider) VideoGeneration(ctx *schemas.BifrostContext, key // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // use handle provider response body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Parse response @@ -2349,7 +2356,7 @@ func (provider *GeminiProvider) VideoRetrieve(ctx *schemas.BifrostContext, key s // Handle error response if resp.StatusCode() != fasthttp.StatusOK { respBody := append([]byte(nil), resp.Body()...) - return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), nil, respBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), nil, respBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Parse response @@ -2438,10 +2445,10 @@ func (provider *GeminiProvider) VideoDownload(ctx *schemas.BifrostContext, key s if resp.StatusCode() != fasthttp.StatusOK { // log full error provider.logger.Error("failed to download video: " + string(resp.Body())) - return nil, providerUtils.NewBifrostOperationError( + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostOperationError( fmt.Sprintf("failed to download video: HTTP %d", resp.StatusCode()), nil, - ) + ), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { @@ -2591,24 +2598,24 @@ func (provider *GeminiProvider) BatchCreate(ctx *schemas.BifrostContext, key sch latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Parse the batch job response var geminiResp GeminiBatchJobResponse if err := sonic.Unmarshal(body, &geminiResp); err != nil { provider.logger.Error("gemini batch create unmarshal error: " + err.Error()) - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseUnmarshal, err), jsonData, body, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseUnmarshal, err), jsonData, body, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Check for metadata if geminiResp.Metadata == nil { @@ -2624,8 +2631,9 @@ func (provider *GeminiProvider) BatchCreate(ctx *schemas.BifrostContext, key sch failedCount := 0 // If results are already available (fast completion), count them - if geminiResp.Dest != nil && len(geminiResp.Dest.InlinedResponses) > 0 { - for _, inlineResp := range geminiResp.Dest.InlinedResponses { + outputFile, inlinedResponses := geminiBatchOutput(&geminiResp) + if len(inlinedResponses) > 0 { + for _, inlineResp := range inlinedResponses { if inlineResp.Error != nil { failedCount++ } else if inlineResp.Response != nil { @@ -2640,9 +2648,9 @@ func (provider *GeminiProvider) BatchCreate(ctx *schemas.BifrostContext, key sch status := ToBifrostBatchStatus(geminiResp.Metadata.State) // If state is empty but we have results, it's completed - if geminiResp.Metadata.State == "" && geminiResp.Dest != nil && len(geminiResp.Dest.InlinedResponses) > 0 { + if geminiResp.Metadata.State == "" && len(inlinedResponses) > 0 { status = schemas.BatchStatusCompleted - completedCount = len(geminiResp.Dest.InlinedResponses) - failedCount + completedCount = len(inlinedResponses) - failedCount } // Build response @@ -2669,8 +2677,8 @@ func (provider *GeminiProvider) BatchCreate(ctx *schemas.BifrostContext, key sch } // Include output file ID if results are in a file - if geminiResp.Dest != nil && geminiResp.Dest.FileName != "" { - result.OutputFileID = &geminiResp.Dest.FileName + if outputFile != "" { + result.OutputFileID = &outputFile } return result, nil @@ -2729,7 +2737,7 @@ func (provider *GeminiProvider) batchListByKey(ctx *schemas.BifrostContext, key }, }, latency, nil } - return nil, latency, parseGeminiError(resp) + return nil, latency, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -2875,7 +2883,7 @@ func (provider *GeminiProvider) batchRetrieveByKey(ctx *schemas.BifrostContext, // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, parseGeminiError(resp) + return nil, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -2899,7 +2907,7 @@ func (provider *GeminiProvider) batchRetrieveByKey(ctx *schemas.BifrostContext, geminiResp.Metadata.State == GeminiBatchStateCancelled || geminiResp.Metadata.State == GeminiBatchStateExpired - return &schemas.BifrostBatchRetrieveResponse{ + result := &schemas.BifrostBatchRetrieveResponse{ ID: geminiResp.Metadata.Name, Object: "batch", Status: ToBifrostBatchStatus(geminiResp.Metadata.State), @@ -2916,7 +2924,16 @@ func (provider *GeminiProvider) batchRetrieveByKey(ctx *schemas.BifrostContext, ExtraFields: schemas.BifrostResponseExtraFields{ Latency: latency.Milliseconds(), }, - }, nil + } + + // Surface the output file id for file-based batches so callers can download the + // results via files.content. Inline batches carry no file; their responses are + // exposed through BatchResults instead. + if outputFile, _ := geminiBatchOutput(&geminiResp); outputFile != "" { + result.OutputFileID = &outputFile + } + + return result, nil } // BatchRetrieve retrieves a specific batch job for Gemini, trying each key until successful. @@ -2986,9 +3003,9 @@ func (provider *GeminiProvider) batchCancelByKey(ctx *schemas.BifrostContext, ke if resp.StatusCode() == fasthttp.StatusNotFound || resp.StatusCode() == fasthttp.StatusMethodNotAllowed { // 404 could mean batch not found or cancel not supported // Return the error instead of assuming completed - return nil, parseGeminiError(resp) + return nil, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } - return nil, parseGeminiError(resp) + return nil, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } now := time.Now().Unix() @@ -3065,7 +3082,7 @@ func (provider *GeminiProvider) batchDeleteByKey(ctx *schemas.BifrostContext, ke } if resp.StatusCode() != fasthttp.StatusOK && resp.StatusCode() != fasthttp.StatusNoContent { - return nil, parseGeminiError(resp) + return nil, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } return &schemas.BifrostBatchDeleteResponse{ @@ -3272,7 +3289,7 @@ func (provider *GeminiProvider) batchResultsByKey(ctx *schemas.BifrostContext, k // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, parseGeminiError(resp) + return nil, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -3297,66 +3314,21 @@ func (provider *GeminiProvider) batchResultsByKey(ctx *schemas.BifrostContext, k var results []schemas.BatchResultItem var parseErrors []schemas.BatchError - if geminiResp.Dest != nil && geminiResp.Dest.FileName != "" { + outputFile, inlinedResponses := geminiBatchOutput(&geminiResp) + if outputFile != "" { // File-based results: download and parse the results file - provider.logger.Debug("gemini batch results in file: " + geminiResp.Dest.FileName) - fileResults, fileParseErrors, bifrostErr := provider.downloadBatchResultsFile(ctx, key, geminiResp.Dest.FileName) + provider.logger.Debug("gemini batch results in file: " + outputFile) + fileResults, fileParseErrors, bifrostErr := provider.downloadBatchResultsFile(ctx, key, outputFile) if bifrostErr != nil { return nil, bifrostErr } results = fileResults parseErrors = fileParseErrors - } else if geminiResp.Dest != nil && len(geminiResp.Dest.InlinedResponses) > 0 { - // Inline results: extract from inlinedResponses - results = make([]schemas.BatchResultItem, 0, len(geminiResp.Dest.InlinedResponses)) - for i, inlineResp := range geminiResp.Dest.InlinedResponses { - customID := fmt.Sprintf("request-%d", i) - if inlineResp.Metadata != nil && inlineResp.Metadata.Key != "" { - customID = inlineResp.Metadata.Key - } - - resultItem := schemas.BatchResultItem{ - CustomID: customID, - } - - if inlineResp.Error != nil { - resultItem.Error = &schemas.BatchResultError{ - Code: fmt.Sprintf("%d", inlineResp.Error.Code), - Message: inlineResp.Error.Message, - } - } else if inlineResp.Response != nil { - // Convert the response to a map for the Body field - respBody := make(map[string]interface{}) - if len(inlineResp.Response.Candidates) > 0 { - candidate := inlineResp.Response.Candidates[0] - if candidate.Content != nil && len(candidate.Content.Parts) > 0 { - var textParts []string - for _, part := range candidate.Content.Parts { - if part.Text != "" { - textParts = append(textParts, part.Text) - } - } - if len(textParts) > 0 { - respBody["text"] = strings.Join(textParts, "") - } - } - respBody["finish_reason"] = string(candidate.FinishReason) - } - if inlineResp.Response.UsageMetadata != nil { - respBody["usage"] = map[string]interface{}{ - "prompt_tokens": inlineResp.Response.UsageMetadata.PromptTokenCount, - "completion_tokens": inlineResp.Response.UsageMetadata.CandidatesTokenCount, - "total_tokens": inlineResp.Response.UsageMetadata.TotalTokenCount, - } - } - - resultItem.Response = &schemas.BatchResultResponse{ - StatusCode: 200, - Body: respBody, - } - } - - results = append(results, resultItem) + } else if len(inlinedResponses) > 0 { + // Inline results: extract from response.inlinedResponses + results = make([]schemas.BatchResultItem, 0, len(inlinedResponses)) + for i, inlineResp := range inlinedResponses { + results = append(results, geminiInlineResponseToBatchResultItem(inlineResp, fmt.Sprintf("request-%d", i))) } } @@ -3389,8 +3361,9 @@ func (provider *GeminiProvider) batchResultsByKey(ctx *schemas.BifrostContext, k } // BatchResults retrieves batch results for Gemini, trying each key until successful. -// Results are extracted from dest.inlinedResponses for inline batches, -// or downloaded from dest.fileName for file-based batches. +// Results are extracted from the batch response's inline responses +// (response.inlinedResponses) for inline batches, or downloaded from the responses +// file (response.responsesFile) for file-based batches. func (provider *GeminiProvider) BatchResults(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostBatchResultsRequest) (*schemas.BifrostBatchResultsResponse, *schemas.BifrostError) { if err := providerUtils.CheckOperationAllowed(schemas.Gemini, provider.customProviderConfig, schemas.BatchResultsRequest); err != nil { return nil, err @@ -3442,6 +3415,16 @@ func (provider *GeminiProvider) FileUpload(ctx *schemas.BifrostContext, key sche if err != nil { return nil, providerUtils.NewBifrostOperationError("failed to marshal metadata", err) } + contentType := "" + if request.ContentType != nil { + contentType = strings.TrimSpace(*request.ContentType) + } + if contentType != "" { + metadataJSON, err = providerUtils.SetJSONField(metadataJSON, "file.mimeType", contentType) + if err != nil { + return nil, providerUtils.NewBifrostOperationError("failed to marshal metadata", err) + } + } if _, err := metadataField.Write(metadataJSON); err != nil { return nil, providerUtils.NewBifrostOperationError("failed to write metadata", err) } @@ -3451,7 +3434,15 @@ func (provider *GeminiProvider) FileUpload(ctx *schemas.BifrostContext, key sche if filename == "" { filename = "file.bin" } - part, err := writer.CreateFormFile("file", filename) + var part io.Writer + if contentType != "" { + header := make(textproto.MIMEHeader) + header.Set("Content-Disposition", multipart.FileContentDisposition("file", filename)) + header.Set("Content-Type", contentType) + part, err = writer.CreatePart(header) + } else { + part, err = writer.CreateFormFile("file", filename) + } if err != nil { return nil, providerUtils.NewBifrostOperationError("failed to create form file", err) } @@ -3491,7 +3482,7 @@ func (provider *GeminiProvider) FileUpload(ctx *schemas.BifrostContext, key sche // Handle error response if resp.StatusCode() != fasthttp.StatusOK && resp.StatusCode() != fasthttp.StatusCreated { - return nil, parseGeminiError(resp) + return nil, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -3583,7 +3574,7 @@ func (provider *GeminiProvider) fileListByKey(ctx *schemas.BifrostContext, key s // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, latency, parseGeminiError(resp) + return nil, latency, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -3746,7 +3737,7 @@ func (provider *GeminiProvider) fileRetrieveByKey(ctx *schemas.BifrostContext, k // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, parseGeminiError(resp) + return nil, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -3859,7 +3850,7 @@ func (provider *GeminiProvider) fileDeleteByKey(ctx *schemas.BifrostContext, key // Handle error response - DELETE returns 200 with empty body on success if resp.StatusCode() != fasthttp.StatusOK && resp.StatusCode() != fasthttp.StatusNoContent { - return nil, parseGeminiError(resp) + return nil, providerUtils.SetErrorLatency(parseGeminiError(resp), latency) } return &schemas.BifrostFileDeleteResponse{ @@ -3985,7 +3976,7 @@ func (provider *GeminiProvider) CountTokens(ctx *schemas.BifrostContext, key sch latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Keep passthrough request mode for countTokens, but fully drain the remaining // client upload before returning. This avoids proxy-layer broken-pipe 502s when @@ -3996,12 +3987,12 @@ func (provider *GeminiProvider) CountTokens(ctx *schemas.BifrostContext, key sch } if resp.StatusCode() != fasthttp.StatusOK { - return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseGeminiError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } responseBody := append([]byte(nil), body...) @@ -4015,7 +4006,7 @@ func (provider *GeminiProvider) CountTokens(ctx *schemas.BifrostContext, key sch providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), ) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } response := geminiResponse.ToBifrostCountTokensResponse(request.Model) @@ -4195,22 +4186,24 @@ func (provider *GeminiProvider) PassthroughStream( fasthttpReq.SetBody(req.Body) activeClient := providerUtils.PrepareResponseStreaming(ctx, provider.streamingClient, resp) - if err := activeClient.Do(fasthttpReq, resp); err != nil { + err := activeClient.Do(fasthttpReq, resp) + latency := time.Since(startTime) + if err != nil { providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } - return nil, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), latency) } headers := providerUtils.ExtractPassthroughProviderResponseHeaders(resp) diff --git a/core/providers/gemini/gemini_test.go b/core/providers/gemini/gemini_test.go index fc5e58a3394..7a5f9568efc 100644 --- a/core/providers/gemini/gemini_test.go +++ b/core/providers/gemini/gemini_test.go @@ -391,6 +391,62 @@ func TestMissingThoughtSignatureUsesBypassSentinel(t *testing.T) { assert.NotContains(t, string(encoded), `"thoughtSignature":"c2tpcF90aG91Z2h0X3NpZ25hdHVyZV92YWxpZGF0b3I="`) } +func TestGeminiChatCompletionRejectsGCSImageURL(t *testing.T) { + _, err := gemini.ToGeminiChatCompletionRequest(nil, &schemas.BifrostChatRequest{ + Model: "gemini-3-flash-preview", + Input: []schemas.ChatMessage{ + { + Role: schemas.ChatMessageRoleUser, + Content: &schemas.ChatMessageContent{ + ContentBlocks: []schemas.ChatContentBlock{ + { + Type: schemas.ChatContentBlockTypeText, + Text: schemas.Ptr("Describe this image."), + }, + { + Type: schemas.ChatContentBlockTypeImage, + ImageURLStruct: &schemas.ChatInputImage{ + URL: "gs://my-bucket/xxx.png", + }, + }, + }, + }, + }, + }, + }) + + require.Error(t, err) + assert.Contains(t, err.Error(), `URL scheme "gs" is not allowed`) +} + +func TestGeminiResponsesRejectsGCSImageURL(t *testing.T) { + _, err := gemini.ToGeminiResponsesRequest(nil, &schemas.BifrostResponsesRequest{ + Model: "gemini-3-flash-preview", + Input: []schemas.ResponsesMessage{ + { + Role: schemas.Ptr(schemas.ResponsesInputMessageRoleUser), + Content: &schemas.ResponsesMessageContent{ + ContentBlocks: []schemas.ResponsesMessageContentBlock{ + { + Type: schemas.ResponsesInputMessageContentBlockTypeText, + Text: schemas.Ptr("Describe this image."), + }, + { + Type: schemas.ResponsesInputMessageContentBlockTypeImage, + ResponsesInputMessageContentBlockImage: &schemas.ResponsesInputMessageContentBlockImage{ + ImageURL: schemas.Ptr("gs://my-bucket/xxx.png"), + }, + }, + }, + }, + }, + }, + }) + + require.Error(t, err) + assert.Contains(t, err.Error(), `URL scheme "gs" is not allowed`) +} + func TestEmbeddedThoughtSignatureDoesNotUseBypassSentinel(t *testing.T) { thoughtSig := base64.RawURLEncoding.EncodeToString([]byte{0x01, 0x02, 0x03}) callID := "call_1_ts_" + thoughtSig @@ -3409,6 +3465,7 @@ func TestThinkingBudgetValidation_Chat(t *testing.T) { wantDisabled bool // budget=0: expect IncludeThoughts=false wantDynamic bool // budget=-1: expect ThinkingBudget=-1 wantBudget *int32 // expected ThinkingBudget value when no error + wantNoConfig bool }{ // gemini-2.5-pro: valid range [128, 32768] { @@ -3495,11 +3552,17 @@ func TestThinkingBudgetValidation_Chat(t *testing.T) { // Special values — exempt from range checks on any model. { - name: "budget_zero_disables_thinking", - model: "gemini-2.5-pro", + name: "flash_budget_zero_disables_thinking", + model: "gemini-2.5-flash", budget: 0, wantDisabled: true, }, + { + name: "pro_budget_zero_omits_thinking_config", + model: "gemini-2.5-pro", + budget: 0, + wantNoConfig: true, + }, { name: "budget_minus_one_dynamic", model: "gemini-2.5-flash", @@ -3529,6 +3592,10 @@ func TestThinkingBudgetValidation_Chat(t *testing.T) { require.NoError(t, err) require.NotNil(t, result) + if tt.wantNoConfig { + assert.Nil(t, result.GenerationConfig.ThinkingConfig) + return + } require.NotNil(t, result.GenerationConfig.ThinkingConfig, "ThinkingConfig should be set") tc := result.GenerationConfig.ThinkingConfig @@ -3559,6 +3626,7 @@ func TestThinkingBudgetValidation_Responses(t *testing.T) { wantDisabled bool wantDynamic bool wantBudget *int32 + wantNoConfig bool }{ // gemini-2.5-pro { @@ -3601,6 +3669,12 @@ func TestThinkingBudgetValidation_Responses(t *testing.T) { budget: 0, wantDisabled: true, }, + { + name: "pro_budget_zero_omits_thinking_config", + model: "gemini-2.5-pro", + budget: 0, + wantNoConfig: true, + }, { name: "budget_minus_one_dynamic", model: "gemini-2.5-pro", @@ -3629,6 +3703,10 @@ func TestThinkingBudgetValidation_Responses(t *testing.T) { require.NoError(t, err) require.NotNil(t, result) + if tt.wantNoConfig { + assert.Nil(t, result.GenerationConfig.ThinkingConfig) + return + } require.NotNil(t, result.GenerationConfig.ThinkingConfig, "ThinkingConfig should be set") tc := result.GenerationConfig.ThinkingConfig @@ -3648,6 +3726,25 @@ func TestThinkingBudgetValidation_Responses(t *testing.T) { } } +func TestThinkingBudgetZeroUnsupportedForProResponses(t *testing.T) { + effortNone := "none" + + req := &schemas.BifrostResponsesRequest{ + Model: "gemini-2.5-pro", + Params: &schemas.ResponsesParameters{ + Reasoning: &schemas.ResponsesParametersReasoning{ + Effort: &effortNone, + }, + }, + } + + result, err := gemini.ToGeminiResponsesRequest(nil, req) + + require.NoError(t, err) + require.NotNil(t, result) + assert.Nil(t, result.GenerationConfig.ThinkingConfig) +} + // TestThinkingBudgetEffortUsesModelRange verifies that effort-based budget // calculation uses the correct model-specific range, not a global default. // In particular, gemini-2.5-flash-lite (min=512) must not use gemini-2.5-flash's @@ -4356,4 +4453,4 @@ func TestImagenImageEditSizeRoundtrip(t *testing.T) { assert.Equal(t, "2K", *imagenReq.Parameters.SampleImageSize) require.NotNil(t, imagenReq.Parameters.AspectRatio, "AspectRatio must be derived from Size on edit path") assert.Equal(t, "1:1", *imagenReq.Parameters.AspectRatio) -} \ No newline at end of file +} diff --git a/core/providers/gemini/responses.go b/core/providers/gemini/responses.go index 3e7486fa0a6..ec27d0b797b 100644 --- a/core/providers/gemini/responses.go +++ b/core/providers/gemini/responses.go @@ -77,6 +77,12 @@ func (request *GeminiGenerationRequest) ToBifrostResponsesRequest(ctx *schemas.B } func ToGeminiResponsesRequest(ctx *schemas.BifrostContext, bifrostReq *schemas.BifrostResponsesRequest) (*GeminiGenerationRequest, error) { + return ToGeminiResponsesRequestWithImageURLSchemes(ctx, bifrostReq, defaultGeminiImageURLSchemes...) +} + +// ToGeminiResponsesRequestWithImageURLSchemes converts a Bifrost Responses request +// to Gemini format using the provider-specific allowlist for non-data image URLs. +func ToGeminiResponsesRequestWithImageURLSchemes(ctx *schemas.BifrostContext, bifrostReq *schemas.BifrostResponsesRequest, allowedImageURLSchemes ...string) (*GeminiGenerationRequest, error) { if bifrostReq == nil { return nil, nil } @@ -106,9 +112,20 @@ func ToGeminiResponsesRequest(ctx *schemas.BifrostContext, bifrostReq *schemas.B return nil, err } - // Convert tool choice if present + // Convert tool choice if present, but only when function declarations exist. + // Gemini rejects functionCallingConfig without function_declarations + // (e.g. a web-search-only request has GoogleSearch but no declarations). if bifrostReq.Params.ToolChoice != nil { - geminiReq.ToolConfig = convertResponsesToolChoiceToGemini(bifrostReq.Params.ToolChoice) + hasFunctionDeclarations := false + for _, tool := range geminiReq.Tools { + if len(tool.FunctionDeclarations) > 0 { + hasFunctionDeclarations = true + break + } + } + if hasFunctionDeclarations { + geminiReq.ToolConfig = convertResponsesToolChoiceToGemini(bifrostReq.Params.ToolChoice) + } } } @@ -119,7 +136,7 @@ func ToGeminiResponsesRequest(ctx *schemas.BifrostContext, bifrostReq *schemas.B // Convert ResponsesInput messages to Gemini contents if bifrostReq.Input != nil { - contents, systemInstruction, err := convertResponsesMessagesToGeminiContents(bifrostReq.Input, capModel, bifrostReq.Provider) + contents, systemInstruction, err := convertResponsesMessagesToGeminiContents(bifrostReq.Input, capModel, bifrostReq.Provider, allowedImageURLSchemes...) if err != nil { return nil, err } @@ -2849,16 +2866,14 @@ func (r *GeminiGenerationRequest) convertParamsToGenerationConfigResponses(param // Handle "none" effort explicitly (only if max_tokens not present) if !hasMaxTokens && hasEffort && *params.Reasoning.Effort == "none" { - config.ThinkingConfig.IncludeThoughts = false - config.ThinkingConfig.ThinkingBudget = schemas.Ptr(int32(0)) + setThinkingBudgetZeroIfSupported(&config, capModel) } else if hasMaxTokens { // User provided max_tokens - use thinkingBudget (all Gemini models support this) // If both max_tokens and effort are present, we ignore effort and use ONLY max_tokens budget := *params.Reasoning.MaxTokens switch budget { case 0: - config.ThinkingConfig.IncludeThoughts = false - config.ThinkingConfig.ThinkingBudget = schemas.Ptr(int32(0)) + setThinkingBudgetZeroIfSupported(&config, capModel) case DynamicReasoningBudget: // Special case: -1 means dynamic budget config.ThinkingConfig.ThinkingBudget = schemas.Ptr(int32(DynamicReasoningBudget)) default: @@ -3049,7 +3064,11 @@ func convertResponsesToolChoiceToGemini(toolChoice *schemas.ResponsesToolChoice) // responses, where a tool returns images/files nested in functionResponse.parts). provider // distinguishes Vertex AI from the Gemini Developer API, which differ in how multimodal // function responses must be referenced (see the FunctionCallOutput handling below). -func convertResponsesMessagesToGeminiContents(messages []schemas.ResponsesMessage, model string, provider schemas.ModelProvider) ([]Content, *Content, error) { +func convertResponsesMessagesToGeminiContents(messages []schemas.ResponsesMessage, model string, provider schemas.ModelProvider, allowedImageURLSchemes ...string) ([]Content, *Content, error) { + if len(allowedImageURLSchemes) == 0 { + allowedImageURLSchemes = defaultGeminiImageURLSchemes + } + isVertex := provider == schemas.Vertex // if only system / developer message is there, convert it to user message (since openai allows it) if len(messages) == 1 && messages[0].Role != nil && (*messages[0].Role == schemas.ResponsesInputMessageRoleSystem || *messages[0].Role == schemas.ResponsesInputMessageRoleDeveloper) { @@ -3062,7 +3081,7 @@ func convertResponsesMessagesToGeminiContents(messages []schemas.ResponsesMessag } if messages[0].Content.ContentBlocks != nil { for _, block := range messages[0].Content.ContentBlocks { - part, err := convertContentBlockToGeminiPart(block) + part, err := convertContentBlockToGeminiPart(block, allowedImageURLSchemes...) if err != nil { return nil, nil, fmt.Errorf("failed to convert system message content block: %w", err) } @@ -3119,7 +3138,7 @@ func convertResponsesMessagesToGeminiContents(messages []schemas.ResponsesMessag } if msg.Content.ContentBlocks != nil { for _, block := range msg.Content.ContentBlocks { - part, err := convertContentBlockToGeminiPart(block) + part, err := convertContentBlockToGeminiPart(block, allowedImageURLSchemes...) if err != nil { return nil, nil, fmt.Errorf("failed to convert system message content block: %w", err) } @@ -3223,6 +3242,10 @@ func convertResponsesMessagesToGeminiContents(messages []schemas.ResponsesMessag } } + if part.ThoughtSignature == nil { + part.ThoughtSignature = []byte(skipThoughtSignatureValidator) + } + content.Parts = append(content.Parts, part) } @@ -3271,7 +3294,7 @@ func convertResponsesMessagesToGeminiContents(messages []schemas.ResponsesMessag if !supportsMultimodalToolOutput { continue // older models can't accept media in a function response } - mediaPart, err := convertContentBlockToGeminiPart(block) + mediaPart, err := convertContentBlockToGeminiPart(block, allowedImageURLSchemes...) if err != nil { return nil, nil, fmt.Errorf("failed to convert function output content block: %w", err) } @@ -3361,7 +3384,7 @@ func convertResponsesMessagesToGeminiContents(messages []schemas.ResponsesMessag if msg.Content.ContentBlocks != nil { for _, block := range msg.Content.ContentBlocks { - part, err := convertContentBlockToGeminiPart(block) + part, err := convertContentBlockToGeminiPart(block, allowedImageURLSchemes...) if err != nil { return nil, nil, fmt.Errorf("failed to convert message content block: %w", err) } @@ -3382,7 +3405,11 @@ func convertResponsesMessagesToGeminiContents(messages []schemas.ResponsesMessag } // convertContentBlockToGeminiPart converts a content block to Gemini part -func convertContentBlockToGeminiPart(block schemas.ResponsesMessageContentBlock) (*Part, error) { +func convertContentBlockToGeminiPart(block schemas.ResponsesMessageContentBlock, allowedImageURLSchemes ...string) (*Part, error) { + if len(allowedImageURLSchemes) == 0 { + allowedImageURLSchemes = defaultGeminiImageURLSchemes + } + switch block.Type { case schemas.ResponsesInputMessageContentBlockTypeText, schemas.ResponsesOutputMessageContentTypeText: @@ -3428,7 +3455,7 @@ func convertContentBlockToGeminiPart(block schemas.ResponsesMessageContentBlock) imageURL := *block.ResponsesInputMessageContentBlockImage.ImageURL // Use existing utility functions to handle URL parsing - sanitizedURL, err := schemas.SanitizeImageURL(imageURL) + sanitizedURL, err := schemas.SanitizeImageURLWithAllowedSchemes(imageURL, allowedImageURLSchemes...) if err != nil { return nil, fmt.Errorf("failed to sanitize image URL: %w", err) } diff --git a/core/providers/gemini/types.go b/core/providers/gemini/types.go index 6802a6a8d33..4646144cb7a 100644 --- a/core/providers/gemini/types.go +++ b/core/providers/gemini/types.go @@ -42,7 +42,7 @@ var thinkingBudgetRanges = []struct { } // thoughtSignatureSeparator is used to separate the base ID from the thought signature in tool IDs -const thoughtSignatureSeparator = "_ts_" +const thoughtSignatureSeparator = providerUtils.ThoughtSignatureSeparator type Role string @@ -2261,7 +2261,8 @@ type GeminiBatchMetadataInputConfig struct { // GeminiBatchMetadataOutputConfig represents the output config in batch job metadata. type GeminiBatchMetadataOutputConfig struct { - ResponsesFile string `json:"responsesFile,omitempty"` + ResponsesFile string `json:"responsesFile,omitempty"` + InlinedResponses *GeminiInlinedResponses `json:"inlinedResponses,omitempty"` } // GeminiBatchMetadata contains metadata for tracking batch requests. @@ -2290,18 +2291,26 @@ type GeminiBatchJobResponse struct { Response *GeminiBatchOutput `json:"response,omitempty"` } -// GeminiBatchOutput represents the output of a successful batch job. +// GeminiBatchOutput represents the output of a successful batch job. It mirrors the +// GenerateContentBatchOutput union returned under the Operation's response field +// (and metadata.output): either a responses file or a set of inline responses. type GeminiBatchOutput struct { - Type string `json:"@type,omitempty"` - ResponsesFile string `json:"responsesFile,omitempty"` + Type string `json:"@type,omitempty"` + ResponsesFile string `json:"responsesFile,omitempty"` + InlinedResponses *GeminiInlinedResponses `json:"inlinedResponses,omitempty"` } -// GeminiBatchDest contains the destination/output of a batch job. -// For inline requests, results are in InlinedResponses. -// For file-based input, results are in a file referenced by FileName. +// GeminiBatchDest is the client-SDK-facing output shape (dest.fileName) emitted when +// converting a Bifrost batch response back to Gemini format. The raw REST API reports +// output under the Operation's response / metadata.output fields, not dest. type GeminiBatchDest struct { + FileName string `json:"fileName,omitempty"` +} + +// GeminiInlinedResponses wraps the array of inline batch responses. The REST API nests +// the array one level deep: response.inlinedResponses.inlinedResponses[]. +type GeminiInlinedResponses struct { InlinedResponses []GeminiInlinedResponse `json:"inlinedResponses,omitempty"` - FileName string `json:"fileName,omitempty"` } // GeminiInlinedResponse represents a single response in the batch output. diff --git a/core/providers/gemini/utils.go b/core/providers/gemini/utils.go index c3961e174d3..0615ba01fb3 100644 --- a/core/providers/gemini/utils.go +++ b/core/providers/gemini/utils.go @@ -18,6 +18,8 @@ import ( "github.com/valyala/fasthttp" ) +var defaultGeminiImageURLSchemes = []string{"http", "https"} + // isGemini3Plus returns true if the model is Gemini 3.0 or higher // Uses simple string operations for hot path performance func isGemini3Plus(model string) bool { @@ -165,6 +167,22 @@ func supportsThinkingConfig(model string) bool { return isGemini3Plus(model) } +func canDisableThinkingWithBudget(model string) bool { + return !strings.Contains(strings.ToLower(model), "gemini-2.5-pro") +} + +func setThinkingBudgetZeroIfSupported(config *GenerationConfig, model string) { + if !canDisableThinkingWithBudget(model) { + config.ThinkingConfig = nil + return + } + if config.ThinkingConfig == nil { + config.ThinkingConfig = &GenerationConfigThinkingConfig{} + } + config.ThinkingConfig.IncludeThoughts = false + config.ThinkingConfig.ThinkingBudget = schemas.Ptr(int32(0)) +} + // effortToThinkingLevel converts reasoning effort to Gemini ThinkingLevel string // Pro models only support "low" or "high" // Other models support "minimal", "low", "medium", and "high" @@ -1184,16 +1202,14 @@ func convertParamsToGenerationConfig(params *schemas.ChatParameters, responseMod // Handle "none" effort explicitly (only if max_tokens not present) if !hasMaxTokens && hasEffort && *params.Reasoning.Effort == "none" { - config.ThinkingConfig.IncludeThoughts = false - config.ThinkingConfig.ThinkingBudget = schemas.Ptr(int32(0)) + setThinkingBudgetZeroIfSupported(&config, model) } else if hasMaxTokens { // User provided max_tokens - use thinkingBudget (all Gemini models support this) // If both max_tokens and effort are present, we ignore effort and use ONLY max_tokens budget := *params.Reasoning.MaxTokens switch budget { case 0: - config.ThinkingConfig.IncludeThoughts = false - config.ThinkingConfig.ThinkingBudget = schemas.Ptr(int32(0)) + setThinkingBudgetZeroIfSupported(&config, model) case DynamicReasoningBudget: // Special case: -1 means dynamic budget config.ThinkingConfig.ThinkingBudget = schemas.Ptr(int32(DynamicReasoningBudget)) default: @@ -1808,12 +1824,16 @@ func addSpeechConfigToGenerationConfig(config *GenerationConfig, voiceConfig *sc } // convertBifrostMessagesToGemini converts Bifrost messages to Gemini format -func convertBifrostMessagesToGemini(messages []schemas.ChatMessage) ([]Content, *Content) { +func convertBifrostMessagesToGemini(messages []schemas.ChatMessage, allowedImageURLSchemes ...string) ([]Content, *Content, error) { + if len(allowedImageURLSchemes) == 0 { + allowedImageURLSchemes = defaultGeminiImageURLSchemes + } + // if only system / developer message is there, convert it to user message (since openai allows it) if len(messages) == 1 && (messages[0].Role == schemas.ChatMessageRoleSystem || messages[0].Role == schemas.ChatMessageRoleDeveloper) { content := convertSystemChatMessageToGeminiUserContent(messages[0]) if len(content.Parts) > 0 { - return []Content{content}, nil + return []Content{content}, nil, nil } } @@ -1999,10 +2019,9 @@ func convertBifrostMessagesToGemini(messages []schemas.ChatMessage) ([]Content, imageURL := block.ImageURLStruct.URL // Sanitize and parse the image URL - sanitizedURL, err := schemas.SanitizeImageURL(imageURL) + sanitizedURL, err := schemas.SanitizeImageURLWithAllowedSchemes(imageURL, allowedImageURLSchemes...) if err != nil { - // Skip this block if URL is invalid - continue + return nil, nil, fmt.Errorf("failed to sanitize image URL: %w", err) } urlInfo := schemas.ExtractURLTypeInfo(sanitizedURL) @@ -2164,7 +2183,7 @@ func convertBifrostMessagesToGemini(messages []schemas.ChatMessage) ([]Content, } } - return contents, systemInstruction + return contents, systemInstruction, nil } func convertSystemChatMessageToGeminiUserContent(message schemas.ChatMessage) Content { diff --git a/core/providers/gemini/videos.go b/core/providers/gemini/videos.go index 8ad4f2738fa..54f0472550a 100644 --- a/core/providers/gemini/videos.go +++ b/core/providers/gemini/videos.go @@ -258,7 +258,7 @@ func ToGeminiVideoGenerationRequest(bifrostReq *schemas.BifrostVideoGenerationRe // Handle input reference (image for image-to-video) if bifrostReq.Input.InputReference != nil && *bifrostReq.Input.InputReference != "" { // extract mime type and base64 string from input reference - sanitizedURL, err := schemas.SanitizeImageURL(*bifrostReq.Input.InputReference) + sanitizedURL, err := schemas.SanitizeImageURLWithAllowedSchemes(*bifrostReq.Input.InputReference, defaultGeminiImageURLSchemes...) if err != nil { return nil, fmt.Errorf("invalid input reference: %w", err) } diff --git a/core/providers/groq/groq.go b/core/providers/groq/groq.go index 5a63cc4e56f..8e5d2c296ed 100644 --- a/core/providers/groq/groq.go +++ b/core/providers/groq/groq.go @@ -104,13 +104,14 @@ func (provider *GroqProvider) ChatCompletion(ctx *schemas.BifrostContext, key sc provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), nil, nil, + nil, provider.logger, ) } @@ -120,17 +121,12 @@ func (provider *GroqProvider) ChatCompletion(ctx *schemas.BifrostContext, key sc // Uses Groq's OpenAI-compatible streaming format. // Returns a channel containing BifrostStreamChunk objects representing the stream or an error if the request fails. func (provider *GroqProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - if v := key.Value.GetValue(); v != "" { - authHeader = map[string]string{"Authorization": "Bearer " + v} - } - // Use shared OpenAI-compatible streaming logic return openai.HandleOpenAIChatCompletionStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+"/v1/chat/completions", request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -142,6 +138,7 @@ func (provider *GroqProvider) ChatCompletionStream(ctx *schemas.BifrostContext, nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) diff --git a/core/providers/huggingface/errors.go b/core/providers/huggingface/errors.go index d98357e0a81..7e2bc83e12e 100644 --- a/core/providers/huggingface/errors.go +++ b/core/providers/huggingface/errors.go @@ -14,15 +14,17 @@ func parseHuggingFaceImageError(resp *fasthttp.Response) *schemas.BifrostError { var errorResp HuggingFaceResponseError bifrostErr := providerUtils.HandleProviderAPIError(resp, &errorResp) - if strings.TrimSpace(errorResp.Type) != "" { - typeCopy := errorResp.Type - bifrostErr.Type = &typeCopy - } - if bifrostErr.Error == nil { bifrostErr.Error = &schemas.ErrorField{} } + if strings.TrimSpace(errorResp.Type) != "" { + bifrostErr.Type = schemas.Ptr(errorResp.Type) + if bifrostErr.Error.Type == nil { + bifrostErr.Error.Type = schemas.Ptr(errorResp.Type) + } + } + // Handle FastAPI validation errors if len(errorResp.Detail) > 0 { var errorMessages []string diff --git a/core/providers/huggingface/errors_test.go b/core/providers/huggingface/errors_test.go new file mode 100644 index 00000000000..88276c1665c --- /dev/null +++ b/core/providers/huggingface/errors_test.go @@ -0,0 +1,27 @@ +package huggingface + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "github.com/valyala/fasthttp" +) + +// TestParseHuggingFaceImageError_PopulatesNestedErrorType verifies the upstream +// exception type is surfaced on the nested error object (error.type), matching +// the other providers, so OpenAI-shaped consumers see it. +func TestParseHuggingFaceImageError_PopulatesNestedErrorType(t *testing.T) { + var resp fasthttp.Response + resp.SetStatusCode(fasthttp.StatusBadRequest) + resp.SetBodyString(`{"type":"validation_error","message":"invalid input"}`) + + bifrostErr := parseHuggingFaceImageError(&resp) + + require.NotNil(t, bifrostErr) + require.NotNil(t, bifrostErr.Error) + require.NotNil(t, bifrostErr.Error.Type, "nested error.type must be populated") + assert.Equal(t, "validation_error", *bifrostErr.Error.Type) + require.NotNil(t, bifrostErr.Type, "top-level type must remain populated") + assert.Equal(t, "validation_error", *bifrostErr.Type) +} diff --git a/core/providers/huggingface/huggingface.go b/core/providers/huggingface/huggingface.go index 28e3359de64..df42640f704 100644 --- a/core/providers/huggingface/huggingface.go +++ b/core/providers/huggingface/huggingface.go @@ -209,10 +209,10 @@ func (provider *HuggingFaceProvider) completeRequestWithModelAliasCache( // Retry the request responseBody, latency, providerResponseHeaders, err = provider.completeRequest(ctx, updatedJSONData, url, key, isHFInferenceAudioRequest, isHFInferenceImageRequest) if err != nil { - return nil, 0, nil, err + return nil, latency, nil, err } } else { - return nil, 0, nil, err + return nil, latency, nil, err } } @@ -257,7 +257,7 @@ func (provider *HuggingFaceProvider) completeRequest(ctx *schemas.BifrostContext // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, latency, providerResponseHeaders, parseHuggingFaceImageError(resp) + return nil, latency, providerResponseHeaders, providerUtils.SetErrorLatency(parseHuggingFaceImageError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -494,7 +494,7 @@ func (provider *HuggingFaceProvider) ChatCompletion(ctx *schemas.BifrostContext, ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse := &schemas.BifrostChatResponse{} @@ -503,7 +503,7 @@ func (provider *HuggingFaceProvider) ChatCompletion(ctx *schemas.BifrostContext, var rawRequest interface{} rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, bifrostResponse, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Ensure model is set correctly @@ -553,11 +553,6 @@ func (provider *HuggingFaceProvider) ChatCompletionStream(ctx *schemas.BifrostCo request.Model = modelName } - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } - customRequestConverter := func(request *schemas.BifrostChatRequest) (providerUtils.RequestBodyWithExtraParams, error) { reqBody, err := ToHuggingFaceChatCompletionRequest(request) if err != nil { @@ -569,13 +564,12 @@ func (provider *HuggingFaceProvider) ChatCompletionStream(ctx *schemas.BifrostCo return reqBody, nil } - // Use shared OpenAI-compatible streaming logic return openai.HandleOpenAIChatCompletionStreaming( ctx, provider.streamingClient, provider.buildRequestURL(ctx, "/v1/chat/completions", schemas.ChatCompletionStreamRequest), request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -587,6 +581,7 @@ func (provider *HuggingFaceProvider) ChatCompletionStream(ctx *schemas.BifrostCo nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -664,7 +659,7 @@ func (provider *HuggingFaceProvider) Embedding(ctx *schemas.BifrostContext, key ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Handle raw request/response for tracking @@ -684,7 +679,7 @@ func (provider *HuggingFaceProvider) Embedding(ctx *schemas.BifrostContext, key // Unmarshal directly to BifrostEmbeddingResponse with custom logic bifrostResponse, convErr := UnmarshalHuggingFaceEmbeddingResponse(responseBody, request.Model) if convErr != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, convErr), jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, convErr), jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Set ExtraFields @@ -746,7 +741,7 @@ func (provider *HuggingFaceProvider) Speech(ctx *schemas.BifrostContext, key sch ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } response := acquireHuggingFaceSpeechResponse() @@ -756,18 +751,18 @@ func (provider *HuggingFaceProvider) Speech(ctx *schemas.BifrostContext, key sch var rawRequest interface{} rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, response, jsonData, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Download the audio file from the URL audioData, downloadErr := provider.downloadAudioFromURL(ctx, response.Audio.URL) if downloadErr != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, downloadErr), jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, downloadErr), jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse, convErr := response.ToBifrostSpeechResponse(request.Model, audioData) if convErr != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, convErr), jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, convErr), jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Set ExtraFields @@ -855,7 +850,7 @@ func (provider *HuggingFaceProvider) Transcription(ctx *schemas.BifrostContext, if err != nil { // Don't wrap raw audio bytes (when isHFInferenceAudioRequest is true) if !isHFInferenceAudioRequest { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } return nil, err } @@ -873,14 +868,14 @@ func (provider *HuggingFaceProvider) Transcription(ctx *schemas.BifrostContext, rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, response, requestBodyForHandling, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { if !isHFInferenceAudioRequest { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } return nil, bifrostErr } bifrostResponse, convErr := response.ToBifrostTranscriptionResponse(request.Model) if convErr != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, convErr), jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, convErr), jsonData, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Set ExtraFields @@ -946,7 +941,7 @@ func (provider *HuggingFaceProvider) ImageGeneration(ctx *schemas.BifrostContext ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Handle raw request/response for tracking @@ -966,7 +961,7 @@ func (provider *HuggingFaceProvider) ImageGeneration(ctx *schemas.BifrostContext // Unmarshal response using Nebius converter bifrostResponse, convErr := UnmarshalHuggingFaceImageGenerationResponse(responseBody, request.Model) if convErr != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, convErr), jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, convErr), jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse.Created = time.Now().Unix() @@ -1064,26 +1059,27 @@ func (provider *HuggingFaceProvider) ImageGenerationStream(ctx *schemas.BifrostC startTime := time.Now() // Make the request err := provider.streamingClient.Do(req, resp) + latency := time.Since(startTime) if err != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } // Request failed before the first response byte (server closed an idle/pooled connection, // broken pipe, connection refused, DNS failure, etc.). Surface as a retriable upstream // connection error (502) so executeRequestWithRetries honors max_retries, matching the // non-streaming path - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), latency) } // Extract provider response headers before status check so error responses also forward them @@ -1092,9 +1088,11 @@ func (provider *HuggingFaceProvider) ImageGenerationStream(ctx *schemas.BifrostC // Check for HTTP errors if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) - return nil, providerUtils.EnrichError(ctx, parseHuggingFaceImageError(resp), jsonBody, nil, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) + return nil, providerUtils.EnrichError(ctx, parseHuggingFaceImageError(resp), jsonBody, nil, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), latency) } + providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) + // Large payload streaming passthrough — pipe raw upstream SSE to client if providerUtils.SetupStreamingPassthrough(ctx, resp) { responseChan := make(chan *schemas.BifrostStreamChunk) @@ -1105,8 +1103,6 @@ func (provider *HuggingFaceProvider) ImageGenerationStream(ctx *schemas.BifrostC // Create response channel responseChan := make(chan *schemas.BifrostStreamChunk, schemas.DefaultStreamBufferSize) - providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) - // Start streaming in a goroutine go func() { defer providerUtils.EnsureStreamFinalizerCalled(ctx, postHookSpanFinalizer) @@ -1320,7 +1316,7 @@ func (provider *HuggingFaceProvider) ImageEdit(ctx *schemas.BifrostContext, key ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) } if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Handle raw request/response for tracking @@ -1340,7 +1336,7 @@ func (provider *HuggingFaceProvider) ImageEdit(ctx *schemas.BifrostContext, key // Unmarshal response bifrostResponse, convErr := UnmarshalHuggingFaceImageGenerationResponse(responseBody, request.Model) if convErr != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, convErr), jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, convErr), jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse.Created = time.Now().Unix() @@ -1447,26 +1443,27 @@ func (provider *HuggingFaceProvider) ImageEditStream(ctx *schemas.BifrostContext startTime := time.Now() // Make the request err := provider.streamingClient.Do(req, resp) + latency := time.Since(startTime) if err != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } // Request failed before the first response byte (server closed an idle/pooled connection, // broken pipe, connection refused, DNS failure, etc.). Surface as a retriable upstream // connection error (502) so executeRequestWithRetries honors max_retries, matching the // non-streaming path - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), latency) } // Extract provider response headers before status check so error responses also forward them @@ -1475,9 +1472,11 @@ func (provider *HuggingFaceProvider) ImageEditStream(ctx *schemas.BifrostContext // Check for HTTP errors if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) - return nil, providerUtils.EnrichError(ctx, parseHuggingFaceImageError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseHuggingFaceImageError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } + providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) + // Large payload streaming passthrough — pipe raw upstream SSE to client if providerUtils.SetupStreamingPassthrough(ctx, resp) { responseChan := make(chan *schemas.BifrostStreamChunk) @@ -1488,8 +1487,6 @@ func (provider *HuggingFaceProvider) ImageEditStream(ctx *schemas.BifrostContext // Create response channel responseChan := make(chan *schemas.BifrostStreamChunk, schemas.DefaultStreamBufferSize) - providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) - // Start streaming in a goroutine go func() { defer providerUtils.EnsureStreamFinalizerCalled(ctx, postHookSpanFinalizer) diff --git a/core/providers/mistral/mistral.go b/core/providers/mistral/mistral.go index 6437a32278e..fe2f5a1f6dd 100644 --- a/core/providers/mistral/mistral.go +++ b/core/providers/mistral/mistral.go @@ -104,7 +104,7 @@ func (provider *MistralProvider) listModelsByKey(ctx *schemas.BifrostContext, ke // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - bifrostErr := ParseMistralError(resp) + bifrostErr := providerUtils.SetErrorLatency(ParseMistralError(resp), latency) return nil, bifrostErr } @@ -180,13 +180,14 @@ func (provider *MistralProvider) ChatCompletion(ctx *schemas.BifrostContext, key provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), provider.normalizeChatRequestForConversion(request), - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), nil, ParseMistralError, + nil, provider.logger, ) } @@ -196,17 +197,12 @@ func (provider *MistralProvider) ChatCompletion(ctx *schemas.BifrostContext, key // Uses Mistral's OpenAI-compatible streaming format. // Returns a channel containing BifrostStreamChunk objects representing the stream or an error if the request fails. func (provider *MistralProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } - // Use shared OpenAI-compatible streaming logic return openai.HandleOpenAIChatCompletionStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+"/v1/chat/completions", provider.normalizeChatRequestForConversion(request), - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -218,6 +214,7 @@ func (provider *MistralProvider) ChatCompletionStream(ctx *schemas.BifrostContex ParseMistralError, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -256,7 +253,7 @@ func (provider *MistralProvider) Embedding(ctx *schemas.BifrostContext, key sche provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/embeddings"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -319,7 +316,7 @@ func (provider *MistralProvider) Transcription(ctx *schemas.BifrostContext, key // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, ParseMistralError(resp) + return nil, providerUtils.SetErrorLatency(ParseMistralError(resp), latency) } responseBody, err := providerUtils.CheckAndDecodeBody(resp) @@ -427,26 +424,27 @@ func (provider *MistralProvider) TranscriptionStream(ctx *schemas.BifrostContext startTime := time.Now() // Make the request err := provider.streamingClient.Do(req, resp) + latency := time.Since(startTime) if err != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } // Request failed before the first response byte (server closed an idle/pooled connection, // broken pipe, connection refused, DNS failure, etc.). Surface as a retriable upstream // connection error (502) so executeRequestWithRetries honors max_retries, matching the // non-streaming path - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), latency) } // Store provider response headers in context before status check so error responses also forward them @@ -455,9 +453,11 @@ func (provider *MistralProvider) TranscriptionStream(ctx *schemas.BifrostContext // Check for HTTP errors if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) - return nil, ParseMistralError(resp) + return nil, providerUtils.SetErrorLatency(ParseMistralError(resp), latency) } + providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) + // Large payload streaming passthrough — pipe raw upstream SSE to client if providerUtils.SetupStreamingPassthrough(ctx, resp) { responseChan := make(chan *schemas.BifrostStreamChunk) @@ -468,8 +468,6 @@ func (provider *MistralProvider) TranscriptionStream(ctx *schemas.BifrostContext // Create response channel responseChan := make(chan *schemas.BifrostStreamChunk, schemas.DefaultStreamBufferSize) - providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) - // Start streaming in a goroutine go func() { defer func() { @@ -670,7 +668,7 @@ func (provider *MistralProvider) OCR(ctx *schemas.BifrostContext, key schemas.Ke // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, ParseMistralError(resp) + return nil, providerUtils.SetErrorLatency(ParseMistralError(resp), latency) } responseBody, err := providerUtils.CheckAndDecodeBody(resp) diff --git a/core/providers/nebius/nebius.go b/core/providers/nebius/nebius.go index d82a4550353..62c4b5898fd 100644 --- a/core/providers/nebius/nebius.go +++ b/core/providers/nebius/nebius.go @@ -92,7 +92,7 @@ func (provider *NebiusProvider) TextCompletion(ctx *schemas.BifrostContext, key provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -107,17 +107,12 @@ func (provider *NebiusProvider) TextCompletion(ctx *schemas.BifrostContext, key // It formats the request, sends it to Nebius, and processes the response. // Returns a channel of BifrostStreamChunk objects or an error if the request fails. func (provider *NebiusProvider) TextCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostTextCompletionRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } - // Use shared OpenAI-compatible streaming logic return openai.HandleOpenAITextCompletionStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/completions"), request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -150,13 +145,14 @@ func (provider *NebiusProvider) ChatCompletion(ctx *schemas.BifrostContext, key provider.client, provider.networkConfig.BaseURL+path, request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), nil, nil, + nil, provider.logger, ) } @@ -166,18 +162,12 @@ func (provider *NebiusProvider) ChatCompletion(ctx *schemas.BifrostContext, key // Uses Nebius's OpenAI-compatible streaming format. // Returns a channel containing BifrostStreamChunk objects representing the stream or an error if the request fails. func (provider *NebiusProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } - - // Use shared OpenAI-compatible streaming logic return openai.HandleOpenAIChatCompletionStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -189,6 +179,7 @@ func (provider *NebiusProvider) ChatCompletionStream(ctx *schemas.BifrostContext nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -226,7 +217,7 @@ func (provider *NebiusProvider) Embedding(ctx *schemas.BifrostContext, key schem provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/embeddings"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -332,12 +323,12 @@ func (provider *NebiusProvider) ImageGeneration(ctx *schemas.BifrostContext, key // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, providerUtils.EnrichError(ctx, parseNebiusImageError(resp), jsonData, nil, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) + return nil, providerUtils.EnrichError(ctx, parseNebiusImageError(resp), jsonData, nil, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } response := &schemas.BifrostImageGenerationResponse{} diff --git a/core/providers/ollama/ollama.go b/core/providers/ollama/ollama.go index ad155200e47..1a641018be8 100644 --- a/core/providers/ollama/ollama.go +++ b/core/providers/ollama/ollama.go @@ -134,7 +134,7 @@ func (provider *OllamaProvider) TextCompletion(ctx *schemas.BifrostContext, key provider.client, baseURL+providerUtils.GetPathFromContext(ctx, "/v1/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -153,16 +153,12 @@ func (provider *OllamaProvider) TextCompletionStream(ctx *schemas.BifrostContext if bifrostErr != nil { return nil, bifrostErr } - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } return openai.HandleOpenAITextCompletionStreaming( ctx, provider.streamingClient, baseURL+providerUtils.GetPathFromContext(ctx, "/v1/completions"), request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -188,13 +184,14 @@ func (provider *OllamaProvider) ChatCompletion(ctx *schemas.BifrostContext, key provider.client, baseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), nil, nil, + nil, provider.logger, ) } @@ -208,17 +205,12 @@ func (provider *OllamaProvider) ChatCompletionStream(ctx *schemas.BifrostContext if bifrostErr != nil { return nil, bifrostErr } - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } - // Use shared OpenAI-compatible streaming logic return openai.HandleOpenAIChatCompletionStreaming( ctx, provider.streamingClient, baseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -230,6 +222,7 @@ func (provider *OllamaProvider) ChatCompletionStream(ctx *schemas.BifrostContext nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -270,7 +263,7 @@ func (provider *OllamaProvider) Embedding(ctx *schemas.BifrostContext, key schem provider.client, baseURL+providerUtils.GetPathFromContext(ctx, "/v1/embeddings"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), diff --git a/core/providers/openai/chat.go b/core/providers/openai/chat.go index 25705096fc0..147fb2dbc27 100644 --- a/core/providers/openai/chat.go +++ b/core/providers/openai/chat.go @@ -10,12 +10,16 @@ import ( // ToBifrostChatRequest converts an OpenAI chat request to Bifrost format func (req *OpenAIChatRequest) ToBifrostChatRequest(ctx *schemas.BifrostContext) *schemas.BifrostChatRequest { provider, model := schemas.ParseModelString(req.Model, "") + params := req.ChatParameters + if params.MaxCompletionTokens == nil && req.MaxTokens != nil { + params.MaxCompletionTokens = req.MaxTokens + } return &schemas.BifrostChatRequest{ Provider: provider, Model: model, Input: ConvertOpenAIMessagesToBifrostMessages(req.Messages), - Params: &req.ChatParameters, + Params: ¶ms, Fallbacks: schemas.ParseFallbacks(req.Fallbacks), } } @@ -63,9 +67,9 @@ func ToOpenAIChatRequest(ctx *schemas.BifrostContext, bifrostReq *schemas.Bifros case schemas.OpenAI, schemas.Azure: openaiReq.normalizeReasoningEffort(capModel) return openaiReq - case schemas.Cerebras: + case schemas.Cerebras, schemas.DeepSeek: openaiReq.filterOpenAISpecificParameters(capModel) - openaiReq.applyCerebrasCompatibility() + openaiReq.stripReasoningDetails() return openaiReq case schemas.XAI: openaiReq.filterOpenAISpecificParameters(capModel) @@ -190,8 +194,9 @@ func (req *OpenAIChatRequest) applyMistralCompatibility() { } } -// applyCerebrasCompatibility applies Cerebras-specific transformations to the request. -func (req *OpenAIChatRequest) applyCerebrasCompatibility() { +// stripReasoningDetails for providers that throw error for reasoning_details in assistant messages +// e.g. Cerebras, DeepSeek +func (req *OpenAIChatRequest) stripReasoningDetails() { for i := range req.Messages { assistantMessage := req.Messages[i].OpenAIChatAssistantMessage if assistantMessage == nil { diff --git a/core/providers/openai/chat_test.go b/core/providers/openai/chat_test.go index d02946ae547..9fafa75b510 100644 --- a/core/providers/openai/chat_test.go +++ b/core/providers/openai/chat_test.go @@ -682,70 +682,83 @@ func TestToOpenAIChatRequest_FireworksPreservesReasoningAndCacheIsolation(t *tes } } -func TestToOpenAIChatRequest_CerebrasStripsAssistantReasoningContent(t *testing.T) { - ctx, cancel := schemas.NewBifrostContextWithCancel(nil) - defer cancel() +func TestToOpenAIChatRequest_StripsAssistantReasoningContentForCompatibleProviders(t *testing.T) { + tests := []struct { + name string + provider schemas.ModelProvider + model string + }{ + {name: "cerebras", provider: schemas.Cerebras, model: "gpt-oss-120b"}, + {name: "deepseek", provider: schemas.DeepSeek, model: "deepseek-v4-pro"}, + } - reasoning := "step by step" - assistantContent := "The weather in Paris is mild today." - userContent := "What is the weather in Paris?" + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + ctx, cancel := schemas.NewBifrostContextWithCancel(nil) + defer cancel() - bifrostReq := &schemas.BifrostChatRequest{ - Provider: schemas.Cerebras, - Model: "gpt-oss-120b", - Input: []schemas.ChatMessage{ - { - Role: schemas.ChatMessageRoleUser, - Content: &schemas.ChatMessageContent{ContentStr: &userContent}, - }, - { - Role: schemas.ChatMessageRoleAssistant, - Content: &schemas.ChatMessageContent{ContentStr: &assistantContent}, - ChatAssistantMessage: &schemas.ChatAssistantMessage{ - Reasoning: &reasoning, + reasoning := "step by step" + assistantContent := "The weather in Paris is mild today." + userContent := "What is the weather in Paris?" + + bifrostReq := &schemas.BifrostChatRequest{ + Provider: tt.provider, + Model: tt.model, + Input: []schemas.ChatMessage{ + { + Role: schemas.ChatMessageRoleUser, + Content: &schemas.ChatMessageContent{ContentStr: &userContent}, + }, + { + Role: schemas.ChatMessageRoleAssistant, + Content: &schemas.ChatMessageContent{ContentStr: &assistantContent}, + ChatAssistantMessage: &schemas.ChatAssistantMessage{ + Reasoning: &reasoning, + }, + }, }, - }, - }, - } + } - result := ToOpenAIChatRequest(ctx, bifrostReq) - if result == nil { - t.Fatal("expected non-nil result") - } - if len(result.Messages) != 2 || result.Messages[1].OpenAIChatAssistantMessage == nil { - t.Fatalf("expected assistant message with OpenAI assistant payload, got %#v", result.Messages) - } - if result.Messages[1].OpenAIChatAssistantMessage.Reasoning != nil { - t.Fatalf("expected assistant reasoning_content to be stripped for cerebras, got %#v", result.Messages[1].OpenAIChatAssistantMessage.Reasoning) - } + result := ToOpenAIChatRequest(ctx, bifrostReq) + if result == nil { + t.Fatal("expected non-nil result") + } + if len(result.Messages) != 2 || result.Messages[1].OpenAIChatAssistantMessage == nil { + t.Fatalf("expected assistant message with OpenAI assistant payload, got %#v", result.Messages) + } + if result.Messages[1].OpenAIChatAssistantMessage.Reasoning != nil { + t.Fatalf("expected assistant reasoning_content to be stripped for %s, got %#v", tt.provider, result.Messages[1].OpenAIChatAssistantMessage.Reasoning) + } - ctx.SetValue(schemas.BifrostContextKeyPassthroughExtraParams, true) - wireBody, bifrostErr := providerUtils.CheckContextAndGetRequestBody( - ctx, - bifrostReq, - func() (providerUtils.RequestBodyWithExtraParams, error) { - return ToOpenAIChatRequest(ctx, bifrostReq), nil - }, - ) - if bifrostErr != nil { - t.Fatalf("failed to build request body: %v", bifrostErr.Error.Message) - } + ctx.SetValue(schemas.BifrostContextKeyPassthroughExtraParams, true) + wireBody, bifrostErr := providerUtils.CheckContextAndGetRequestBody( + ctx, + bifrostReq, + func() (providerUtils.RequestBodyWithExtraParams, error) { + return ToOpenAIChatRequest(ctx, bifrostReq), nil + }, + ) + if bifrostErr != nil { + t.Fatalf("failed to build request body: %v", bifrostErr.Error.Message) + } - var jsonMap map[string]interface{} - if err := sonic.Unmarshal(wireBody, &jsonMap); err != nil { - t.Fatalf("failed to parse marshaled request body: %v", err) - } + var jsonMap map[string]any + if err := sonic.Unmarshal(wireBody, &jsonMap); err != nil { + t.Fatalf("failed to parse marshaled request body: %v", err) + } - messages, ok := jsonMap["messages"].([]interface{}) - if !ok || len(messages) != 2 { - t.Fatalf("expected 2 messages in wire payload, got %#v", jsonMap["messages"]) - } - assistantMessage, ok := messages[1].(map[string]interface{}) - if !ok { - t.Fatalf("expected assistant message object, got %#v", messages[1]) - } - if _, ok := assistantMessage["reasoning_content"]; ok { - t.Fatalf("expected reasoning_content to be absent from cerebras assistant payload, got %#v", assistantMessage["reasoning_content"]) + messages, ok := jsonMap["messages"].([]any) + if !ok || len(messages) != 2 { + t.Fatalf("expected 2 messages in wire payload, got %#v", jsonMap["messages"]) + } + assistantMessage, ok := messages[1].(map[string]any) + if !ok { + t.Fatalf("expected assistant message object, got %#v", messages[1]) + } + if _, ok := assistantMessage["reasoning_content"]; ok { + t.Fatalf("expected reasoning_content to be absent from %s assistant payload, got %#v", tt.provider, assistantMessage["reasoning_content"]) + } + }) } } @@ -1208,7 +1221,7 @@ func TestToOpenAIChatRequest_CacheControl_OpenRouterOnly(t *testing.T) { // (sonic.Unmarshal of the raw body into *OpenAIChatRequest, then // ToBifrostChatRequest) and asserts the top-level server-tool name survives. func TestOpenAIInbound_ServerToolNameSurvives(t *testing.T) { - body := `{"model":"bedrock/global.anthropic.claude-sonnet-4-6","max_tokens":1024,"tools":[{"type":"bash_20250124","name":"bash"}],"messages":[{"role":"user","content":"Run ls"}]}` + body := `{"model":"bedrock/global.anthropic.claude-sonnet-4-6","max_tokens":8000,"tools":[{"type":"bash_20250124","name":"bash"}],"messages":[{"role":"user","content":"Run ls"}]}` var req OpenAIChatRequest if err := sonic.Unmarshal([]byte(body), &req); err != nil { @@ -1224,4 +1237,131 @@ func TestOpenAIInbound_ServerToolNameSurvives(t *testing.T) { if bifReq.Params == nil || len(bifReq.Params.Tools) != 1 || bifReq.Params.Tools[0].Name != "bash" { t.Fatalf("ToBifrostChatRequest dropped name: %+v", bifReq.Params) } + if bifReq.Params.MaxCompletionTokens == nil || *bifReq.Params.MaxCompletionTokens != 8000 { + t.Fatalf("ToBifrostChatRequest did not map max_tokens to max_completion_tokens: %+v", bifReq.Params.MaxCompletionTokens) + } +} + +func TestOpenAIInbound_MaxCompletionTokensTakesPriorityOverMaxTokens(t *testing.T) { + body := `{"model":"bedrock/global.anthropic.claude-sonnet-4-6","max_tokens":100,"max_completion_tokens":200,"messages":[{"role":"user","content":"Run ls"}]}` + + var req OpenAIChatRequest + if err := sonic.Unmarshal([]byte(body), &req); err != nil { + t.Fatalf("unmarshal: %v", err) + } + + ctx := schemas.NewBifrostContext(nil, schemas.NoDeadline) + bifReq := req.ToBifrostChatRequest(ctx) + if bifReq.Params == nil || bifReq.Params.MaxCompletionTokens == nil { + t.Fatalf("ToBifrostChatRequest dropped max_completion_tokens: %+v", bifReq.Params) + } + if *bifReq.Params.MaxCompletionTokens != 200 { + t.Fatalf("max_completion_tokens should take priority over max_tokens, got %d", *bifReq.Params.MaxCompletionTokens) + } +} + +// When a conversation switches from Gemini to OpenAI, Gemini's thoughtSignature is +// embedded in the tool call_id as "_ts_" and can exceed OpenAI's 64-char +// limit. The chat converter must strip it to the base ID on the wire while leaving the +// caller's input intact (so a later Gemini turn can still recover the signature). +func TestToOpenAIChatRequest_StripsThoughtSignatureFromToolCallIDs(t *testing.T) { + embeddedID := "search" + providerUtils.ThoughtSignatureSeparator + strings.Repeat("A", 6000) + + req := &schemas.BifrostChatRequest{ + Provider: schemas.OpenAI, + Model: "gpt-4o", + Input: []schemas.ChatMessage{ + { + Role: schemas.ChatMessageRoleAssistant, + ChatAssistantMessage: &schemas.ChatAssistantMessage{ + ToolCalls: []schemas.ChatAssistantMessageToolCall{{ + ID: schemas.Ptr(embeddedID), + Type: schemas.Ptr("function"), + Function: schemas.ChatAssistantMessageToolCallFunction{ + Name: schemas.Ptr("search"), + Arguments: "{}", + }, + }}, + }, + }, + { + Role: schemas.ChatMessageRoleTool, + ChatToolMessage: &schemas.ChatToolMessage{ToolCallID: schemas.Ptr(embeddedID)}, + Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("result")}, + }, + }, + } + + ctx, cancel := schemas.NewBifrostContextWithCancel(nil) + defer cancel() + result := ToOpenAIChatRequest(ctx, req) + require.NotNil(t, result) + + gotCallID := *result.Messages[0].OpenAIChatAssistantMessage.ToolCalls[0].ID + gotToolCallID := *result.Messages[1].ChatToolMessage.ToolCallID + + if gotCallID != "search" { + t.Errorf("assistant tool call ID: got %q, want %q", gotCallID, "search") + } + if len(gotCallID) > 64 { + t.Errorf("assistant tool call ID exceeds OpenAI's 64-char limit: %d chars", len(gotCallID)) + } + if gotToolCallID != gotCallID { + t.Errorf("tool result ID %q must match assistant call ID %q", gotToolCallID, gotCallID) + } + + // The caller's history must be untouched. + if *req.Input[0].ChatAssistantMessage.ToolCalls[0].ID != embeddedID { + t.Error("original assistant tool call ID was mutated") + } + if *req.Input[1].ChatToolMessage.ToolCallID != embeddedID { + t.Error("original tool result tool_call_id was mutated") + } +} + +// A short call id that merely contains "_ts_" (e.g. two distinct raw upstream ids) must be +// left intact: stripping only kicks in above OpenAI's 64-char limit, so distinct ids never +// collapse into one. +func TestToOpenAIChatRequest_PreservesShortToolCallIDsContainingSeparator(t *testing.T) { + req := &schemas.BifrostChatRequest{ + Provider: schemas.OpenAI, + Model: "gpt-4o", + Input: []schemas.ChatMessage{ + { + Role: schemas.ChatMessageRoleAssistant, + ChatAssistantMessage: &schemas.ChatAssistantMessage{ + ToolCalls: []schemas.ChatAssistantMessageToolCall{ + { + ID: schemas.Ptr("search_ts_a"), + Type: schemas.Ptr("function"), + Function: schemas.ChatAssistantMessageToolCallFunction{Name: schemas.Ptr("search"), Arguments: "{}"}, + }, + { + ID: schemas.Ptr("search_ts_b"), + Type: schemas.Ptr("function"), + Function: schemas.ChatAssistantMessageToolCallFunction{Name: schemas.Ptr("search"), Arguments: "{}"}, + }, + }, + }, + }, + { + Role: schemas.ChatMessageRoleTool, + ChatToolMessage: &schemas.ChatToolMessage{ToolCallID: schemas.Ptr("search_ts_a")}, + Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("r")}, + }, + }, + } + + ctx, cancel := schemas.NewBifrostContextWithCancel(nil) + defer cancel() + result := ToOpenAIChatRequest(ctx, req) + require.NotNil(t, result) + + got := result.Messages[0].OpenAIChatAssistantMessage.ToolCalls + if *got[0].ID != "search_ts_a" || *got[1].ID != "search_ts_b" { + t.Errorf("distinct short ids must be preserved, got %q and %q", *got[0].ID, *got[1].ID) + } + if *result.Messages[1].ChatToolMessage.ToolCallID != "search_ts_a" { + t.Errorf("short tool_call_id must be preserved, got %q", *result.Messages[1].ChatToolMessage.ToolCallID) + } } diff --git a/core/providers/openai/large_payload.go b/core/providers/openai/large_payload.go index fe3aaf18128..ae1ec528d8c 100644 --- a/core/providers/openai/large_payload.go +++ b/core/providers/openai/large_payload.go @@ -39,7 +39,7 @@ func handleOpenAILargePayloadPassthrough( ctx *schemas.BifrostContext, client *fasthttp.Client, url string, - key schemas.Key, + authHeader map[string]string, extraHeaders map[string]string, providerName schemas.ModelProvider, logger schemas.Logger, @@ -57,8 +57,8 @@ func handleOpenAILargePayloadPassthrough( providerUtils.SetExtraHeaders(ctx, req, extraHeaders, nil) req.SetRequestURI(url) req.Header.SetMethod(http.MethodPost) - if key.Value.GetValue() != "" { - req.Header.Set("Authorization", "Bearer "+key.Value.GetValue()) + for k, v := range authHeader { + req.Header.Set(k, v) } // Rewrite model prefix and stream request body to upstream. diff --git a/core/providers/openai/openai.go b/core/providers/openai/openai.go index 39bc527e141..4230bf7c6a1 100644 --- a/core/providers/openai/openai.go +++ b/core/providers/openai/openai.go @@ -170,7 +170,7 @@ func ListModelsByKey( // Handle error response if resp.StatusCode() != fasthttp.StatusOK { bifrostErr := ParseOpenAIError(resp) - return nil, bifrostErr + return nil, providerUtils.SetErrorLatency(bifrostErr, latency) } // Copy response body before releasing @@ -181,7 +181,7 @@ func ListModelsByKey( // Use enhanced response handler with pre-allocated response rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, openaiResponse, nil, sendBackRawRequest, sendBackRawResponse) if bifrostErr != nil { - return nil, bifrostErr + return nil, providerUtils.SetErrorLatency(bifrostErr, latency) } response := openaiResponse.ToBifrostListModelsResponse(providerName, key.Models, key.BlacklistedModels, key.Aliases, unfiltered) @@ -202,6 +202,18 @@ func ListModelsByKey( return response, nil } +// BearerAuthHeader builds the auth header map for OpenAI-compatible providers that authenticate +// with a bearer token. It returns an empty (non-nil) map when the key carries no value (e.g. +// SigV4 / header-based auth supplied via extraHeaders), so callers can pass it directly to the +// Handle*Request functions that take an authHeader map. +func BearerAuthHeader(key schemas.Key) map[string]string { + headers := map[string]string{} + if key.Value.GetValue() != "" { + headers["Authorization"] = "Bearer " + key.Value.GetValue() + } + return headers +} + // HandleOpenAIListModelsRequest handles a list models request to OpenAI's API. func HandleOpenAIListModelsRequest( ctx *schemas.BifrostContext, @@ -239,7 +251,7 @@ func (provider *OpenAIProvider) TextCompletion(ctx *schemas.BifrostContext, key provider.client, provider.buildRequestURL(ctx, "/v1/completions", schemas.TextCompletionRequest), request, - key, + BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -256,7 +268,7 @@ func HandleOpenAITextCompletionRequest( client *fasthttp.Client, url string, request *schemas.BifrostTextCompletionRequest, - key schemas.Key, + authHeader map[string]string, extraHeaders map[string]string, providerName schemas.ModelProvider, sendBackRawRequest bool, @@ -285,12 +297,12 @@ func HandleOpenAITextCompletionRequest( req.Header.SetMethod(http.MethodPost) req.Header.SetContentType("application/json") - if key.Value.GetValue() != "" { - req.Header.Set("Authorization", "Bearer "+key.Value.GetValue()) + for k, v := range authHeader { + req.Header.Set(k, v) } // Large payload passthrough: stream body directly without JSON marshaling - if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, key, extraHeaders, providerName, logger); handled { + if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, authHeader, extraHeaders, providerName, logger); handled { if lpErr != nil { return nil, lpErr } @@ -325,7 +337,7 @@ func HandleOpenAITextCompletionRequest( latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Extract provider response headers early so they're available on error paths too providerResponseHeaders := providerUtils.ExtractProviderResponseHeaders(resp) @@ -335,15 +347,15 @@ func HandleOpenAITextCompletionRequest( if resp.StatusCode() != fasthttp.StatusOK { providerUtils.MaterializeStreamErrorBody(ctx, resp) if customErrorConverter != nil { - return nil, providerUtils.EnrichError(ctx, customErrorConverter(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, customErrorConverter(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } body, lpResult, finalErr := finalizeOpenAIResponse(ctx, resp, latency, providerName, logger) respOwned = false // ownership transferred if finalErr != nil { - return nil, providerUtils.EnrichError(ctx, finalErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, finalErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } if lpResult != nil { return &schemas.BifrostTextCompletionResponse{ @@ -364,7 +376,7 @@ func HandleOpenAITextCompletionRequest( } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, sendBackRawRequest, sendBackRawResponse, latency) } response.ExtraFields.Latency = latency.Milliseconds() @@ -390,16 +402,12 @@ func (provider *OpenAIProvider) TextCompletionStream(ctx *schemas.BifrostContext if err := providerUtils.CheckOperationAllowed(schemas.OpenAI, provider.customProviderConfig, schemas.TextCompletionStreamRequest); err != nil { return nil, err } - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } return HandleOpenAITextCompletionStreaming( ctx, provider.streamingClient, provider.buildRequestURL(ctx, "/v1/completions", schemas.TextCompletionStreamRequest), request, - authHeader, + BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -492,6 +500,7 @@ func HandleOpenAITextCompletionStreaming( err := activeClient.Do(req, resp) if err != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) + latency := time.Since(startTime) if errors.Is(err, context.Canceled) { return nil, providerUtils.EnrichError(ctx, &schemas.BifrostError{ IsBifrostError: false, @@ -500,10 +509,10 @@ func HandleOpenAITextCompletionStreaming( Message: schemas.ErrRequestCancelled, Error: err, }, - }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // The request failed before the first response byte (connection refused, server // closed an idle/pooled connection, broken pipe, DNS failure, etc.). Mirror the @@ -512,7 +521,7 @@ func HandleOpenAITextCompletionStreaming( // (500, IsBifrostError=true). The latter caused the retry loop in executeRequestWithRetries // to break early on IsBifrostError, so max_retries never applied to streaming connection // failures - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Store provider response headers in context before status check so error responses also forward them @@ -522,12 +531,15 @@ func HandleOpenAITextCompletionStreaming( if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) providerUtils.MaterializeStreamErrorBody(ctx, resp) + latency := time.Since(startTime) if customErrorConverter != nil { - return nil, providerUtils.EnrichError(ctx, customErrorConverter(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, customErrorConverter(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } + latency := time.Since(startTime) + // Large payload streaming passthrough — pipe raw upstream SSE to client if providerUtils.SetupStreamingPassthrough(ctx, resp) { responseChan := make(chan *schemas.BifrostStreamChunk) @@ -615,7 +627,7 @@ func HandleOpenAITextCompletionStreaming( handlerErr.ExtraFields.RawResponse = rawResponse } ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, handlerErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, handlerErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } } else { @@ -627,7 +639,7 @@ func HandleOpenAITextCompletionStreaming( if err := sonic.UnmarshalString(jsonData, &bifrostErr); err == nil { if bifrostErr.Error != nil && bifrostErr.Error.Message != "" { ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, &bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, &bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } } @@ -760,13 +772,14 @@ func (provider *OpenAIProvider) ChatCompletion(ctx *schemas.BifrostContext, key provider.client, provider.buildRequestURL(ctx, "/v1/chat/completions", schemas.ChatCompletionRequest), request, - key, + BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), nil, nil, + nil, provider.logger, ) } @@ -777,13 +790,14 @@ func HandleOpenAIChatCompletionRequest( client *fasthttp.Client, url string, request *schemas.BifrostChatRequest, - key schemas.Key, + authHeader map[string]string, extraHeaders map[string]string, sendBackRawRequest bool, sendBackRawResponse bool, providerName schemas.ModelProvider, customResponseHandler responseHandler[schemas.BifrostChatResponse], customErrorConverter ErrorConverter, + signer providerUtils.BodySigner, logger schemas.Logger, ) (*schemas.BifrostChatResponse, *schemas.BifrostError) { // Create request @@ -806,12 +820,12 @@ func HandleOpenAIChatCompletionRequest( req.Header.SetMethod(http.MethodPost) req.Header.SetContentType("application/json") - if key.Value.GetValue() != "" { - req.Header.Set("Authorization", "Bearer "+key.Value.GetValue()) + for k, v := range authHeader { + req.Header.Set(k, v) } // Large payload passthrough: stream body directly without JSON marshaling - if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, key, extraHeaders, providerName, logger); handled { + if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, authHeader, extraHeaders, providerName, logger); handled { if lpErr != nil { return nil, lpErr } @@ -840,13 +854,23 @@ func HandleOpenAIChatCompletionRequest( return nil, bifrostErr } + if signer != nil { + sigHeaders, bErr := signer(jsonData) + if bErr != nil { + return nil, bErr + } + for k, v := range sigHeaders { + req.Header.Set(k, v) + } + } + req.SetBody(jsonData) // Make request latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Extract provider response headers early so they're available on error paths too providerResponseHeaders := providerUtils.ExtractProviderResponseHeaders(resp) @@ -857,15 +881,15 @@ func HandleOpenAIChatCompletionRequest( providerUtils.MaterializeStreamErrorBody(ctx, resp) logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) if customErrorConverter != nil { - return nil, providerUtils.EnrichError(ctx, customErrorConverter(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, customErrorConverter(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } body, lpResult, finalErr := finalizeOpenAIResponse(ctx, resp, latency, providerName, logger) respOwned = false // ownership transferred if finalErr != nil { - return nil, providerUtils.EnrichError(ctx, finalErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, finalErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } if lpResult != nil { return &schemas.BifrostChatResponse{ @@ -886,7 +910,7 @@ func HandleOpenAIChatCompletionRequest( } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, sendBackRawRequest, sendBackRawResponse, latency) } response.ExtraFields.Latency = latency.Milliseconds() @@ -912,10 +936,6 @@ func (provider *OpenAIProvider) ChatCompletionStream(ctx *schemas.BifrostContext if err := providerUtils.CheckOperationAllowed(schemas.OpenAI, provider.customProviderConfig, schemas.ChatCompletionStreamRequest); err != nil { return nil, err } - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } if provider.disableStore { if request.Params == nil { request.Params = &schemas.ChatParameters{} @@ -929,7 +949,7 @@ func (provider *OpenAIProvider) ChatCompletionStream(ctx *schemas.BifrostContext provider.streamingClient, provider.buildRequestURL(ctx, "/v1/chat/completions", schemas.ChatCompletionStreamRequest), request, - authHeader, + BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -941,6 +961,7 @@ func (provider *OpenAIProvider) ChatCompletionStream(ctx *schemas.BifrostContext nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -965,6 +986,7 @@ func HandleOpenAIChatCompletionStreaming( customErrorConverter ErrorConverter, postRequestConverter func(*OpenAIChatRequest) *OpenAIChatRequest, postResponseConverter func(*schemas.BifrostChatResponse) *schemas.BifrostChatResponse, + signer providerUtils.BodySigner, logger schemas.Logger, postHookSpanFinalizer func(context.Context), ) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { @@ -1033,6 +1055,17 @@ func HandleOpenAIChatCompletionStreaming( req.Header.Set(key, value) } + if signer != nil { + sigHeaders, bErr := signer(jsonBody) + if bErr != nil { + defer providerUtils.ReleaseStreamingResponse(ctx, resp) + return nil, bErr + } + for k, v := range sigHeaders { + req.Header.Set(k, v) + } + } + setStreamingRequestBody(ctx, req, jsonBody, providerName) // Use streaming-aware client when large payload optimization is active — ensures @@ -1042,6 +1075,7 @@ func HandleOpenAIChatCompletionStreaming( startTime := time.Now() // Make the request err := activeClient.Do(req, resp) + latency := time.Since(startTime) if err != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { @@ -1052,10 +1086,10 @@ func HandleOpenAIChatCompletionStreaming( Message: schemas.ErrRequestCancelled, Error: err, }, - }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // The request failed before the first response byte (connection refused, server // closed an idle/pooled connection, broken pipe, DNS failure, etc.). Mirror the @@ -1064,7 +1098,7 @@ func HandleOpenAIChatCompletionStreaming( // (500, IsBifrostError=true). The latter caused the retry loop in executeRequestWithRetries // to break early on IsBifrostError, so max_retries never applied to streaming connection // failures - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Store provider response headers in context before status check so error responses also forward them @@ -1075,9 +1109,9 @@ func HandleOpenAIChatCompletionStreaming( defer providerUtils.ReleaseStreamingResponse(ctx, resp) providerUtils.MaterializeStreamErrorBody(ctx, resp) if customErrorConverter != nil { - return nil, providerUtils.EnrichError(ctx, customErrorConverter(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, customErrorConverter(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Large payload streaming passthrough — pipe raw upstream SSE to client @@ -1172,7 +1206,7 @@ func HandleOpenAIChatCompletionStreaming( if err := sonic.UnmarshalString(jsonData, &bifrostErr); err == nil { if bifrostErr.Error != nil && bifrostErr.Error.Message != "" { ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, &bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, &bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } } @@ -1190,7 +1224,7 @@ func HandleOpenAIChatCompletionStreaming( handlerErr.ExtraFields.RawResponse = rawResponse } ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, handlerErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, handlerErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } } else { @@ -1252,7 +1286,7 @@ func HandleOpenAIChatCompletionStreaming( } ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } @@ -1428,13 +1462,14 @@ func (provider *OpenAIProvider) Responses(ctx *schemas.BifrostContext, key schem provider.client, provider.buildRequestURL(ctx, "/v1/responses", schemas.ResponsesRequest), request, - key, + BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), nil, nil, + nil, provider.logger, ) } @@ -1445,13 +1480,14 @@ func HandleOpenAIResponsesRequest( client *fasthttp.Client, url string, request *schemas.BifrostResponsesRequest, - key schemas.Key, + authHeader map[string]string, extraHeaders map[string]string, sendBackRawRequest bool, sendBackRawResponse bool, providerName schemas.ModelProvider, customResponseHandler responseHandler[schemas.BifrostResponsesResponse], customErrorConverter ErrorConverter, + signer providerUtils.BodySigner, logger schemas.Logger, ) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) { // Create request @@ -1474,12 +1510,12 @@ func HandleOpenAIResponsesRequest( req.Header.SetMethod(http.MethodPost) req.Header.SetContentType("application/json") - if key.Value.GetValue() != "" { - req.Header.Set("Authorization", "Bearer "+key.Value.GetValue()) + for k, v := range authHeader { + req.Header.Set(k, v) } // Large payload passthrough: stream body directly without JSON marshaling - if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, key, extraHeaders, providerName, logger); handled { + if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, authHeader, extraHeaders, providerName, logger); handled { if lpErr != nil { return nil, lpErr } @@ -1508,13 +1544,23 @@ func HandleOpenAIResponsesRequest( return nil, bifrostErr } + if signer != nil { + sigHeaders, bErr := signer(jsonData) + if bErr != nil { + return nil, bErr + } + for k, v := range sigHeaders { + req.Header.Set(k, v) + } + } + req.SetBody(jsonData) // Make request latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Extract provider response headers early so they're available on error paths too providerResponseHeaders := providerUtils.ExtractProviderResponseHeaders(resp) @@ -1525,15 +1571,15 @@ func HandleOpenAIResponsesRequest( providerUtils.MaterializeStreamErrorBody(ctx, resp) logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) if customErrorConverter != nil { - return nil, providerUtils.EnrichError(ctx, customErrorConverter(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, customErrorConverter(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } body, lpResult, finalErr := finalizeOpenAIResponse(ctx, resp, latency, providerName, logger) respOwned = false // ownership transferred if finalErr != nil { - return nil, providerUtils.EnrichError(ctx, finalErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, finalErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } if lpResult != nil { return &schemas.BifrostResponsesResponse{ @@ -1553,7 +1599,7 @@ func HandleOpenAIResponsesRequest( } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, sendBackRawRequest, sendBackRawResponse, latency) } response.ExtraFields.Latency = latency.Milliseconds() @@ -1583,10 +1629,6 @@ func (provider *OpenAIProvider) ResponsesStream(ctx *schemas.BifrostContext, pos if err := providerUtils.CheckOperationAllowed(schemas.OpenAI, provider.customProviderConfig, schemas.ResponsesStreamRequest); err != nil { return nil, err } - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } if provider.disableStore { if request.Params == nil { request.Params = &schemas.ResponsesParameters{} @@ -1600,7 +1642,7 @@ func (provider *OpenAIProvider) ResponsesStream(ctx *schemas.BifrostContext, pos provider.streamingClient, provider.buildRequestURL(ctx, "/v1/responses", schemas.ResponsesStreamRequest), request, - authHeader, + BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -1611,6 +1653,7 @@ func (provider *OpenAIProvider) ResponsesStream(ctx *schemas.BifrostContext, pos nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -1634,6 +1677,7 @@ func HandleOpenAIResponsesStreaming( customErrorConverter ErrorConverter, postRequestConverter func(*OpenAIResponsesRequest) *OpenAIResponsesRequest, postResponseConverter func(*schemas.BifrostResponsesStreamResponse) *schemas.BifrostResponsesStreamResponse, + signer providerUtils.BodySigner, logger schemas.Logger, postHookSpanFinalizer func(context.Context), ) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { @@ -1685,6 +1729,17 @@ func HandleOpenAIResponsesStreaming( req.Header.Set(key, value) } + if signer != nil { + sigHeaders, bErr := signer(jsonBody) + if bErr != nil { + defer providerUtils.ReleaseStreamingResponse(ctx, resp) + return nil, bErr + } + for k, v := range sigHeaders { + req.Header.Set(k, v) + } + } + setStreamingRequestBody(ctx, req, jsonBody, providerName) // Use streaming-aware client when large payload optimization is active — ensures @@ -1694,6 +1749,7 @@ func HandleOpenAIResponsesStreaming( startTime := time.Now() // Make the request err := activeClient.Do(req, resp) + latency := time.Since(startTime) if err != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { @@ -1704,10 +1760,10 @@ func HandleOpenAIResponsesStreaming( Message: schemas.ErrRequestCancelled, Error: err, }, - }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // The request failed before the first response byte (connection refused, server // closed an idle/pooled connection, broken pipe, DNS failure, etc.). Mirror the @@ -1716,7 +1772,7 @@ func HandleOpenAIResponsesStreaming( // (500, IsBifrostError=true). The latter caused the retry loop in executeRequestWithRetries // to break early on IsBifrostError, so max_retries never applied to streaming connection // failures - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Store provider response headers in context before status check so error responses also forward them @@ -1727,9 +1783,9 @@ func HandleOpenAIResponsesStreaming( defer providerUtils.ReleaseStreamingResponse(ctx, resp) providerUtils.MaterializeStreamErrorBody(ctx, resp) if customErrorConverter != nil { - return nil, providerUtils.EnrichError(ctx, customErrorConverter(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, customErrorConverter(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Large payload streaming passthrough — pipe raw upstream SSE to client @@ -1812,7 +1868,7 @@ func HandleOpenAIResponsesStreaming( bifrostErr.ExtraFields.RawResponse = rawResponse } ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } } else { @@ -1867,7 +1923,7 @@ func HandleOpenAIResponsesStreaming( } ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, []byte(jsonData), sendBackRawRequest, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, []byte(jsonData), sendBackRawRequest, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } @@ -1884,7 +1940,7 @@ func HandleOpenAIResponsesStreaming( bifrostErr.Error.Code = &response.Response.Error.Code } ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, []byte(jsonData), sendBackRawRequest, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, []byte(jsonData), sendBackRawRequest, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } @@ -1926,7 +1982,7 @@ func (provider *OpenAIProvider) Embedding(ctx *schemas.BifrostContext, key schem provider.client, provider.buildRequestURL(ctx, "/v1/embeddings", schemas.EmbeddingRequest), request, - key, + BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -1943,7 +1999,7 @@ func HandleOpenAIEmbeddingRequest( client *fasthttp.Client, url string, request *schemas.BifrostEmbeddingRequest, - key schemas.Key, + authHeader map[string]string, extraHeaders map[string]string, providerName schemas.ModelProvider, sendBackRawRequest bool, @@ -1971,12 +2027,12 @@ func HandleOpenAIEmbeddingRequest( req.Header.SetMethod(http.MethodPost) req.Header.SetContentType("application/json") - if key.Value.GetValue() != "" { - req.Header.Set("Authorization", "Bearer "+key.Value.GetValue()) + for k, v := range authHeader { + req.Header.Set(k, v) } // Large payload passthrough: stream body directly without JSON marshaling - if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, key, extraHeaders, providerName, logger); handled { + if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, authHeader, extraHeaders, providerName, logger); handled { if lpErr != nil { return nil, lpErr } @@ -2012,7 +2068,7 @@ func HandleOpenAIEmbeddingRequest( latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Extract provider response headers early so they're available on error paths too providerResponseHeaders := providerUtils.ExtractProviderResponseHeaders(resp) @@ -2022,13 +2078,13 @@ func HandleOpenAIEmbeddingRequest( if resp.StatusCode() != fasthttp.StatusOK { providerUtils.MaterializeStreamErrorBody(ctx, resp) logger.Debug(fmt.Sprintf("error from %s provider: %s", providerName, string(resp.Body()))) - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } body, lpResult, finalErr := finalizeOpenAIResponse(ctx, resp, latency, providerName, logger) respOwned = false // ownership transferred if finalErr != nil { - return nil, providerUtils.EnrichError(ctx, finalErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, finalErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } if lpResult != nil { return &schemas.BifrostEmbeddingResponse{ @@ -2049,7 +2105,7 @@ func HandleOpenAIEmbeddingRequest( } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, sendBackRawRequest, sendBackRawResponse, latency) } response.ExtraFields.Latency = latency.Milliseconds() @@ -2141,7 +2197,7 @@ func HandleOpenAISpeechRequest( } // Large payload passthrough: stream body directly without JSON marshaling - if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, key, extraHeaders, providerName, logger); handled { + if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, BearerAuthHeader(key), extraHeaders, providerName, logger); handled { if lpErr != nil { return nil, lpErr } @@ -2166,7 +2222,7 @@ func HandleOpenAISpeechRequest( latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Extract provider response headers early so they're available on error paths too providerResponseHeaders := providerUtils.ExtractProviderResponseHeaders(resp) @@ -2176,14 +2232,14 @@ func HandleOpenAISpeechRequest( if resp.StatusCode() != fasthttp.StatusOK { providerUtils.MaterializeStreamErrorBody(ctx, resp) logger.Debug(fmt.Sprintf("error from %s provider: %s", providerName, string(resp.Body()))) - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Get the binary audio data from the response body body, lpResult, finalErr := finalizeOpenAIResponse(ctx, resp, latency, providerName, logger) respOwned = false // ownership transferred if finalErr != nil { - return nil, providerUtils.EnrichError(ctx, finalErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, finalErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } if lpResult != nil { return &schemas.BifrostSpeechResponse{ @@ -2223,17 +2279,12 @@ func (provider *OpenAIProvider) SpeechStream(ctx *schemas.BifrostContext, postHo } } - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } - return HandleOpenAISpeechStreamRequest( ctx, provider.streamingClient, provider.buildRequestURL(ctx, "/v1/audio/speech", schemas.SpeechStreamRequest), request, - authHeader, + BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -2323,6 +2374,7 @@ func HandleOpenAISpeechStreamRequest( startTime := time.Now() // Make the request err := activeClient.Do(req, resp) + latency := time.Since(startTime) if err != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { @@ -2333,10 +2385,10 @@ func HandleOpenAISpeechStreamRequest( Message: schemas.ErrRequestCancelled, Error: err, }, - }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + }, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // The request failed before the first response byte (connection refused, server // closed an idle/pooled connection, broken pipe, DNS failure, etc.). Mirror the @@ -2345,7 +2397,7 @@ func HandleOpenAISpeechStreamRequest( // (500, IsBifrostError=true). The latter caused the retry loop in executeRequestWithRetries // to break early on IsBifrostError, so max_retries never applied to streaming connection // failures - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Store provider response headers in context before status check so error responses also forward them @@ -2355,7 +2407,7 @@ func HandleOpenAISpeechStreamRequest( if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) providerUtils.MaterializeStreamErrorBody(ctx, resp) - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Large payload streaming passthrough — pipe raw upstream SSE to client @@ -2434,7 +2486,7 @@ func HandleOpenAISpeechStreamRequest( if err := sonic.UnmarshalString(jsonData, &bifrostErr); err == nil { if bifrostErr.Error != nil && bifrostErr.Error.Message != "" { ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, &bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, &bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } } @@ -2520,7 +2572,7 @@ func HandleOpenAITranscriptionRequest( logger schemas.Logger, ) (*schemas.BifrostTranscriptionResponse, *schemas.BifrostError) { // Large payload passthrough: stream multipart body directly without parsing - if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, key, extraHeaders, providerName, logger); handled { + if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, BearerAuthHeader(key), extraHeaders, providerName, logger); handled { if lpErr != nil { return nil, lpErr } @@ -2580,7 +2632,7 @@ func HandleOpenAITranscriptionRequest( latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, bifrostErr + return nil, providerUtils.SetErrorLatency(bifrostErr, latency) } // Extract provider response headers early so they're available on error paths too providerResponseHeaders := providerUtils.ExtractProviderResponseHeaders(resp) @@ -2590,13 +2642,13 @@ func HandleOpenAITranscriptionRequest( if resp.StatusCode() != fasthttp.StatusOK { providerUtils.MaterializeStreamErrorBody(ctx, resp) logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } responseBody, lpResult, finalErr := finalizeOpenAIResponse(ctx, resp, latency, providerName, logger) respOwned = false // ownership transferred if finalErr != nil { - return nil, finalErr + return nil, providerUtils.SetErrorLatency(finalErr, latency) } if lpResult != nil { return &schemas.BifrostTranscriptionResponse{ @@ -2607,12 +2659,12 @@ func HandleOpenAITranscriptionRequest( // Check for empty response trimmed := strings.TrimSpace(string(responseBody)) if len(trimmed) == 0 { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: true, Error: &schemas.ErrorField{ Message: schemas.ErrProviderResponseEmpty, }, - } + }, latency) } copiedResponseBody := append([]byte(nil), responseBody...) @@ -2631,15 +2683,15 @@ func HandleOpenAITranscriptionRequest( if err := sonic.Unmarshal(copiedResponseBody, response); err != nil { // Check if it's an HTML response if providerUtils.IsHTMLResponse(resp, copiedResponseBody) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Message: schemas.ErrProviderResponseHTML, Error: errors.New(string(copiedResponseBody)), }, - } + }, latency) } - return nil, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseUnmarshal, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseUnmarshal, err), latency) } // TODO: add HandleProviderResponse here @@ -2647,13 +2699,13 @@ func HandleOpenAITranscriptionRequest( // Parse raw response for RawResponse field if sendBackRawResponse { if err := sonic.Unmarshal(copiedResponseBody, &rawResponse); err != nil { - return nil, providerUtils.NewBifrostOperationError(schemas.ErrProviderRawResponseUnmarshal, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostOperationError(schemas.ErrProviderRawResponseUnmarshal, err), latency) } } } if bifrostErr != nil { - return nil, bifrostErr + return nil, providerUtils.SetErrorLatency(bifrostErr, latency) } response.ExtraFields = schemas.BifrostResponseExtraFields{ @@ -2674,17 +2726,12 @@ func (provider *OpenAIProvider) TranscriptionStream(ctx *schemas.BifrostContext, return nil, err } - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } - return HandleOpenAITranscriptionStreamRequest( ctx, provider.streamingClient, provider.buildRequestURL(ctx, "/v1/audio/transcriptions", schemas.TranscriptionStreamRequest), request, - authHeader, + BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), @@ -2772,22 +2819,23 @@ func HandleOpenAITranscriptionStreamRequest( startTime := time.Now() // Make the request err := client.Do(req, resp) + latency := time.Since(startTime) if err != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } - return nil, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, err), latency) } // Store provider response headers in context before status check so error responses also forward them @@ -2797,7 +2845,7 @@ func HandleOpenAITranscriptionStreamRequest( if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) providerUtils.MaterializeStreamErrorBody(ctx, resp) - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } // Large payload streaming passthrough — pipe raw upstream SSE to client @@ -2879,7 +2927,7 @@ func HandleOpenAITranscriptionStreamRequest( bifrostErr.ExtraFields.RawResponse = jsonData } ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, body.Bytes(), []byte(jsonData), false, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, body.Bytes(), []byte(jsonData), false, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } } else { @@ -2891,7 +2939,7 @@ func HandleOpenAITranscriptionStreamRequest( if bifrostErrVal.Error != nil && bifrostErrVal.Error.Message != "" { ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) respBody := append([]byte(nil), resp.Body()...) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, &bifrostErrVal, body.Bytes(), respBody, false, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, &bifrostErrVal, body.Bytes(), respBody, false, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } } @@ -3006,7 +3054,7 @@ func HandleOpenAIImageGenerationRequest( } // Large payload passthrough: stream body directly without JSON marshaling - if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, key, extraHeaders, providerName, logger); handled { + if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, BearerAuthHeader(key), extraHeaders, providerName, logger); handled { if lpErr != nil { return nil, lpErr } @@ -3040,7 +3088,7 @@ func HandleOpenAIImageGenerationRequest( latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Extract provider response headers early so they're available on error paths too providerResponseHeaders := providerUtils.ExtractProviderResponseHeaders(resp) @@ -3050,7 +3098,7 @@ func HandleOpenAIImageGenerationRequest( if resp.StatusCode() != fasthttp.StatusOK { providerUtils.MaterializeStreamErrorBody(ctx, resp) logger.Debug(fmt.Sprintf("error from %s provider: %s", providerName, string(resp.Body()))) - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } body, lpResult, finalErr := finalizeOpenAIResponse(ctx, resp, latency, providerName, logger) @@ -3107,17 +3155,13 @@ func (provider *OpenAIProvider) ImageGenerationStream( return nil, err } - var authHeader map[string]string - if value := key.Value.GetValue(); value != "" { - authHeader = map[string]string{"Authorization": "Bearer " + value} - } // Use shared streaming logic return HandleOpenAIImageGenerationStreaming( ctx, provider.streamingClient, provider.buildRequestURL(ctx, "/v1/images/generations", schemas.ImageGenerationStreamRequest), request, - authHeader, + BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -3211,22 +3255,23 @@ func HandleOpenAIImageGenerationStreaming( startTime := time.Now() // Make the request err := activeClient.Do(req, resp) + latency := time.Since(startTime) if err != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } - return nil, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, err), latency) } // Store provider response headers in context before status check so error responses also forward them @@ -3236,7 +3281,7 @@ func HandleOpenAIImageGenerationStreaming( if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) providerUtils.MaterializeStreamErrorBody(ctx, resp) - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Large payload streaming passthrough — pipe raw upstream SSE to client @@ -3286,7 +3331,6 @@ func HandleOpenAIImageGenerationStreaming( sseReader := providerUtils.GetSSEDataReader(ctx, reader) lastChunkTime := startTime - var collectedUsage *schemas.ImageUsage // Track chunk indices per image - similar to how speech/transcription track chunkIndex imageChunkIndices := make(map[int]int) // image index -> chunk index // Track images that have started (via partial chunks) but not yet completed @@ -3320,7 +3364,7 @@ func HandleOpenAIImageGenerationStreaming( if err := sonic.UnmarshalString(jsonData, &bifrostErr); err == nil { if bifrostErr.Error != nil && bifrostErr.Error.Message != "" { ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, &bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, &bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } } @@ -3357,15 +3401,6 @@ func HandleOpenAIImageGenerationStreaming( return } - // Collect usage from completed event - if response.Usage != nil { - collectedUsage = &schemas.ImageUsage{ - InputTokens: response.Usage.InputTokens, - OutputTokens: response.Usage.OutputTokens, - TotalTokens: response.Usage.TotalTokens, - } - } - // Determine if this is the final chunk isCompleted := response.Type == schemas.ImageGenerationEventTypeCompleted @@ -3472,16 +3507,15 @@ func HandleOpenAIImageGenerationStreaming( } if isCompleted { - if collectedUsage != nil { - // Set NImages based on maximum image index seen (maxImageIndex + 1 since indices are 0-based) - if maxImageIndex >= 0 { - nImages := maxImageIndex + 1 - collectedUsage.OutputTokensDetails = &schemas.ImageTokenDetails{ - NImages: nImages, - } + if response.Usage != nil && maxImageIndex >= 0 { + if response.Usage.OutputTokensDetails == nil { + response.Usage.OutputTokensDetails = &schemas.ImageTokenDetails{} + } + if response.Usage.OutputTokensDetails.NImages == 0 { + response.Usage.OutputTokensDetails.NImages = maxImageIndex + 1 } - chunk.Usage = collectedUsage } + chunk.Usage = response.Usage // For completed chunk, use total latency from start chunk.ExtraFields.Latency = time.Since(startTime).Milliseconds() chunk.BackfillParams(&schemas.BifrostRequest{ @@ -3615,7 +3649,7 @@ func (provider *OpenAIProvider) VideoDownload(ctx *schemas.BifrostContext, key s // Handle error response if resp.StatusCode() != fasthttp.StatusOK { provider.logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -3752,7 +3786,7 @@ func HandleOpenAIVideoGenerationRequest( // Handle error response if resp.StatusCode() != fasthttp.StatusOK { logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } responseBody, err := providerUtils.CheckAndDecodeBody(resp) @@ -3847,7 +3881,7 @@ func HandleOpenAIVideoRetrieveRequest( if resp.StatusCode() != fasthttp.StatusOK { logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -3945,7 +3979,7 @@ func HandleOpenAIVideoDeleteRequest( // Handle error response if resp.StatusCode() != fasthttp.StatusOK { logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -4036,7 +4070,7 @@ func HandleOpenAIVideoListRequest( // Handle error response if resp.StatusCode() != fasthttp.StatusOK { logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -4110,7 +4144,7 @@ func (provider *OpenAIProvider) Compaction(ctx *schemas.BifrostContext, key sche provider.client, provider.buildRequestURL(ctx, "/v1/responses/compact", schemas.CompactionRequest), request, - key, + BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), @@ -4125,7 +4159,7 @@ func HandleOpenAICompactionRequest( client *fasthttp.Client, url string, request *schemas.BifrostCompactionRequest, - key schemas.Key, + authHeader map[string]string, extraHeaders map[string]string, sendBackRawRequest bool, sendBackRawResponse bool, @@ -4148,11 +4182,11 @@ func HandleOpenAICompactionRequest( req.Header.SetMethod(http.MethodPost) req.Header.SetContentType("application/json") - if key.Value.GetValue() != "" { - req.Header.Set("Authorization", "Bearer "+key.Value.GetValue()) + for k, v := range authHeader { + req.Header.Set(k, v) } - if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, key, extraHeaders, providerName, logger); handled { + if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, authHeader, extraHeaders, providerName, logger); handled { if lpErr != nil { return nil, lpErr } @@ -4184,7 +4218,7 @@ func HandleOpenAICompactionRequest( latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } providerResponseHeaders := providerUtils.ExtractProviderResponseHeaders(resp) @@ -4193,13 +4227,13 @@ func HandleOpenAICompactionRequest( if resp.StatusCode() != fasthttp.StatusOK { providerUtils.MaterializeStreamErrorBody(ctx, resp) logger.Debug("error from %s provider with status %d", providerName, resp.StatusCode()) - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } body, lpResult, finalErr := finalizeOpenAIResponse(ctx, resp, latency, providerName, logger) respOwned = false if finalErr != nil { - return nil, providerUtils.EnrichError(ctx, finalErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, finalErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } if lpResult != nil { return &schemas.BifrostCompactionResponse{ @@ -4210,7 +4244,7 @@ func HandleOpenAICompactionRequest( response := &schemas.BifrostCompactionResponse{} rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(body, response, jsonData, sendBackRawRequest, sendBackRawResponse) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, sendBackRawRequest, sendBackRawResponse, latency) } response.ExtraFields.Latency = latency.Milliseconds() @@ -4264,7 +4298,7 @@ func HandleOpenAICountTokensRequest( } // Large payload passthrough: stream body directly without JSON marshaling - if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, key, extraHeaders, providerName, logger); handled { + if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, BearerAuthHeader(key), extraHeaders, providerName, logger); handled { if lpErr != nil { return nil, lpErr } @@ -4297,7 +4331,7 @@ func HandleOpenAICountTokensRequest( latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Extract provider response headers early so they're available on error paths too providerResponseHeaders := providerUtils.ExtractProviderResponseHeaders(resp) @@ -4307,7 +4341,7 @@ func HandleOpenAICountTokensRequest( if resp.StatusCode() != fasthttp.StatusOK { providerUtils.MaterializeStreamErrorBody(ctx, resp) logger.Debug(fmt.Sprintf("error from %s provider: %s", providerName, string(resp.Body()))) - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } body, lpResult, finalErr := finalizeOpenAIResponse(ctx, resp, latency, providerName, logger) @@ -4377,7 +4411,7 @@ func HandleOpenAIImageEditRequest( logger schemas.Logger, ) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) { // Large payload passthrough: stream multipart body directly without parsing - if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, key, extraHeaders, providerName, logger); handled { + if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, BearerAuthHeader(key), extraHeaders, providerName, logger); handled { if lpErr != nil { return nil, lpErr } @@ -4435,7 +4469,7 @@ func HandleOpenAIImageEditRequest( latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, nil, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, nil, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Extract provider response headers early so they're available on error paths too providerResponseHeaders := providerUtils.ExtractProviderResponseHeaders(resp) @@ -4443,7 +4477,7 @@ func HandleOpenAIImageEditRequest( if resp.StatusCode() != fasthttp.StatusOK { providerUtils.MaterializeStreamErrorBody(ctx, resp) - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), nil, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), nil, nil, sendBackRawRequest, sendBackRawResponse, latency) } bodyBytes, lpResult, finalErr := finalizeOpenAIResponse(ctx, resp, latency, providerName, logger) @@ -4484,17 +4518,12 @@ func (provider *OpenAIProvider) ImageEditStream(ctx *schemas.BifrostContext, pos return nil, err } - var authHeader map[string]string - if value := key.Value.GetValue(); value != "" { - authHeader = map[string]string{"Authorization": "Bearer " + value} - } - return HandleOpenAIImageEditStreamRequest( ctx, provider.streamingClient, provider.buildRequestURL(ctx, "/v1/images/edits", schemas.ImageEditStreamRequest), request, - authHeader, + BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, false, @@ -4578,22 +4607,23 @@ func HandleOpenAIImageEditStreamRequest( startTime := time.Now() // Make the request err := client.Do(req, resp) + latency := time.Since(startTime) if err != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } - return nil, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, err), latency) } // Store provider response headers in context before status check so error responses also forward them ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerUtils.ExtractProviderResponseHeaders(resp)) @@ -4602,7 +4632,7 @@ func HandleOpenAIImageEditStreamRequest( if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) providerUtils.MaterializeStreamErrorBody(ctx, resp) - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), nil, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), nil, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Large payload streaming passthrough — pipe raw upstream SSE to client @@ -4652,7 +4682,6 @@ func HandleOpenAIImageEditStreamRequest( sseReader := providerUtils.GetSSEDataReader(ctx, reader) lastChunkTime := startTime - var collectedUsage *schemas.ImageUsage // Track chunk indices per image - similar to how speech/transcription track chunkIndex imageChunkIndices := make(map[int]int) // image index -> chunk index // Track images that have started (via partial chunks) but not yet completed @@ -4687,7 +4716,7 @@ func HandleOpenAIImageEditStreamRequest( if err := sonic.UnmarshalString(jsonData, &bifrostErr); err == nil { if bifrostErr.Error != nil && bifrostErr.Error.Message != "" { ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, &bifrostErr, nil, nil, sendBackRawRequest, sendBackRawResponse), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, &bifrostErr, nil, nil, sendBackRawRequest, sendBackRawResponse, latency), responseChan, logger, postHookSpanFinalizer) return } } @@ -4724,15 +4753,6 @@ func HandleOpenAIImageEditStreamRequest( return } - // Collect usage from completed event - if response.Usage != nil { - collectedUsage = &schemas.ImageUsage{ - InputTokens: response.Usage.InputTokens, - OutputTokens: response.Usage.OutputTokens, - TotalTokens: response.Usage.TotalTokens, - } - } - // Determine if this is the final chunk isCompleted := response.Type == schemas.ImageGenerationEventTypeCompleted || response.Type == schemas.ImageEditEventTypeCompleted @@ -4839,16 +4859,15 @@ func HandleOpenAIImageEditStreamRequest( } if isCompleted { - if collectedUsage != nil { - // Set NImages based on maximum image index seen (maxImageIndex + 1 since indices are 0-based) - if maxImageIndex >= 0 { - nImages := maxImageIndex + 1 - collectedUsage.OutputTokensDetails = &schemas.ImageTokenDetails{ - NImages: nImages, - } + if response.Usage != nil && maxImageIndex >= 0 { + if response.Usage.OutputTokensDetails == nil { + response.Usage.OutputTokensDetails = &schemas.ImageTokenDetails{} + } + if response.Usage.OutputTokensDetails.NImages == 0 { + response.Usage.OutputTokensDetails.NImages = maxImageIndex + 1 } - chunk.Usage = collectedUsage } + chunk.Usage = response.Usage // For completed chunk, use total latency from start chunk.ExtraFields.Latency = time.Since(startTime).Milliseconds() chunk.BackfillParams(&schemas.BifrostRequest{ @@ -4906,7 +4925,7 @@ func HandleOpenAIImageVariationRequest( logger schemas.Logger, ) (*schemas.BifrostImageGenerationResponse, *schemas.BifrostError) { // Large payload passthrough: stream multipart body directly without parsing - if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, key, extraHeaders, providerName, logger); handled { + if lpResult, lpErr, handled := handleOpenAILargePayloadPassthrough(ctx, client, url, BearerAuthHeader(key), extraHeaders, providerName, logger); handled { if lpErr != nil { return nil, lpErr } @@ -4963,7 +4982,7 @@ func HandleOpenAIImageVariationRequest( latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, nil, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, nil, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Extract provider response headers early so they're available on error paths too providerResponseHeaders := providerUtils.ExtractProviderResponseHeaders(resp) @@ -4971,7 +4990,7 @@ func HandleOpenAIImageVariationRequest( if resp.StatusCode() != fasthttp.StatusOK { providerUtils.MaterializeStreamErrorBody(ctx, resp) - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), nil, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), nil, nil, sendBackRawRequest, sendBackRawResponse, latency) } bodyBytes, lpResult, finalErr := finalizeOpenAIResponse(ctx, resp, latency, providerName, logger) @@ -5078,7 +5097,7 @@ func (provider *OpenAIProvider) FileUpload(ctx *schemas.BifrostContext, key sche // Handle error response if resp.StatusCode() != fasthttp.StatusOK { provider.logger.Debug("error from %s provider: %s", provider.GetProviderKey(), string(resp.Body())) - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -5173,7 +5192,7 @@ func (provider *OpenAIProvider) FileList(ctx *schemas.BifrostContext, keys []sch // Handle error response if resp.StatusCode() != fasthttp.StatusOK { provider.logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } body, decodeErr := providerUtils.CheckAndDecodeBody(resp) @@ -5529,12 +5548,12 @@ func (provider *OpenAIProvider) VideoRemix(ctx *schemas.BifrostContext, key sche // Handle error response if resp.StatusCode() != fasthttp.StatusOK { provider.logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Parse OpenAI's video response @@ -5648,23 +5667,23 @@ func (provider *OpenAIProvider) BatchCreate(ctx *schemas.BifrostContext, key sch latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } var openAIResp OpenAIBatchResponse rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(body, &openAIResp, jsonData, sendBackRawRequest, sendBackRawResponse) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, body, sendBackRawRequest, sendBackRawResponse, latency) } return openAIResp.ToBifrostBatchCreateResponse(latency, sendBackRawRequest, sendBackRawResponse, rawRequest, rawResponse), nil @@ -5737,7 +5756,7 @@ func (provider *OpenAIProvider) BatchList(ctx *schemas.BifrostContext, keys []sc // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } body, decodeErr := providerUtils.CheckAndDecodeBody(resp) @@ -6134,7 +6153,7 @@ func (provider *OpenAIProvider) ContainerCreate(ctx *schemas.BifrostContext, key // Handle error response if resp.StatusCode() != fasthttp.StatusOK && resp.StatusCode() != fasthttp.StatusCreated { - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } // Parse response @@ -6263,7 +6282,7 @@ func (provider *OpenAIProvider) ContainerList(ctx *schemas.BifrostContext, keys // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } // Parse response @@ -6585,7 +6604,7 @@ func (provider *OpenAIProvider) ContainerFileCreate(ctx *schemas.BifrostContext, // Handle error response if resp.StatusCode() >= 400 { - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } // Decode response body (handles content-encoding like gzip) @@ -6720,7 +6739,7 @@ func (provider *OpenAIProvider) ContainerFileList(ctx *schemas.BifrostContext, k } if resp.StatusCode() >= 400 { - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } // Decode response body (handles content-encoding like gzip) @@ -7218,22 +7237,24 @@ func (provider *OpenAIProvider) PassthroughStream( startTime := time.Now() - if err := activeClient.Do(fasthttpReq, resp); err != nil { + err := activeClient.Do(fasthttpReq, resp) + latency := time.Since(startTime) + if err != nil { providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } - return nil, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, err), latency) } headers := providerUtils.ExtractPassthroughProviderResponseHeaders(resp) diff --git a/core/providers/openai/openai_test.go b/core/providers/openai/openai_test.go index b145a8acc14..bd37cdea6c2 100644 --- a/core/providers/openai/openai_test.go +++ b/core/providers/openai/openai_test.go @@ -95,6 +95,7 @@ func TestOpenAI(t *testing.T) { FileContent: true, FileBatchInput: true, CountTokens: true, + ResponsesLifecycle: true, ExternalCompaction: true, ChatAudio: true, StructuredOutputs: true, // Structured outputs with nullable enum support diff --git a/core/providers/openai/realtime.go b/core/providers/openai/realtime.go index 65cddb4dd30..f58adf2ce66 100644 --- a/core/providers/openai/realtime.go +++ b/core/providers/openai/realtime.go @@ -114,7 +114,7 @@ func (provider *OpenAIProvider) exchangeWebRTCSDP( } req.SetBody(bodyBuf.Bytes()) - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { return "", bifrostErr @@ -122,7 +122,7 @@ func (provider *OpenAIProvider) exchangeWebRTCSDP( answerBody := resp.Body() if resp.StatusCode() < fasthttp.StatusOK || resp.StatusCode() >= fasthttp.StatusMultipleChoices { - return "", provider.realtimeWebRTCUpstreamError(ctx, resp.StatusCode(), answerBody) + return "", providerUtils.SetErrorLatency(provider.realtimeWebRTCUpstreamError(ctx, resp.StatusCode(), answerBody), latency) } return string(answerBody), nil @@ -245,7 +245,7 @@ func (provider *OpenAIProvider) CreateRealtimeClientSecret( ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, headers) if resp.StatusCode() < fasthttp.StatusOK || resp.StatusCode() >= fasthttp.StatusMultipleChoices { - return nil, ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(ParseOpenAIError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) diff --git a/core/providers/openai/responses.go b/core/providers/openai/responses.go index 3ab57be9658..152fcc5a3c4 100644 --- a/core/providers/openai/responses.go +++ b/core/providers/openai/responses.go @@ -92,6 +92,20 @@ func ToOpenAIResponsesRequest(ctx *schemas.BifrostContext, bifrostReq *schemas.B } } + // Strip provider reasoning signatures (e.g. Gemini thoughtSignatures smuggled into + // call_id as "_ts_") from tool call IDs, but only when the id exceeds + // OpenAI's limit — shorter IDs are left intact so distinct upstream IDs are preserved. + // Deterministic, so a call and its output still match. Clone first — the + // ResponsesToolMessage pointer is shared with the caller's input. + if message.ResponsesToolMessage != nil && message.ResponsesToolMessage.CallID != nil && + len(*message.ResponsesToolMessage.CallID) > MaxToolCallIDLength { + if stripped := utils.StripThoughtSignature(*message.ResponsesToolMessage.CallID); stripped != *message.ResponsesToolMessage.CallID { + toolMsgCopy := *message.ResponsesToolMessage + toolMsgCopy.CallID = &stripped + message.ResponsesToolMessage = &toolMsgCopy + } + } + if message.ResponsesReasoning != nil { isGptOss := strings.Contains(capModel, "gpt-oss") isReasoning := isOpenAIReasoningModel(capModel) diff --git a/core/providers/openai/responses_test.go b/core/providers/openai/responses_test.go index 999cb277bb4..65151ab0b8b 100644 --- a/core/providers/openai/responses_test.go +++ b/core/providers/openai/responses_test.go @@ -2018,3 +2018,65 @@ func TestToOpenAIResponsesRequest_OpenRouterServerToolsPreserved(t *testing.T) { } }) } + +// Reverse-direction guard for the Responses path: a Gemini thoughtSignature embedded in +// call_id ("_ts_") must be stripped to the base ID before reaching OpenAI, +// which rejects input[].id over 64 chars. The call and its output strip identically so +// they still pair, and the caller's input is left intact. +func TestToOpenAIResponsesRequest_StripsThoughtSignatureFromCallID(t *testing.T) { + // "_ts_" is the separator used by the native Gemini converters to embed signatures. + embeddedID := "search_ts_" + strings.Repeat("A", 6000) + + req := &schemas.BifrostResponsesRequest{ + Provider: schemas.OpenAI, + Model: "gpt-4o", + Input: []schemas.ResponsesMessage{ + { + Type: schemas.Ptr(schemas.ResponsesMessageTypeFunctionCall), + ResponsesToolMessage: &schemas.ResponsesToolMessage{ + CallID: schemas.Ptr(embeddedID), + Name: schemas.Ptr("search"), + Arguments: schemas.Ptr("{}"), + }, + }, + { + Type: schemas.Ptr(schemas.ResponsesMessageTypeFunctionCallOutput), + ResponsesToolMessage: &schemas.ResponsesToolMessage{ + CallID: schemas.Ptr(embeddedID), + Output: &schemas.ResponsesToolMessageOutputStruct{ + ResponsesToolCallOutputStr: schemas.Ptr("result"), + }, + }, + }, + }, + } + + ctx, cancel := schemas.NewBifrostContextWithCancel(nil) + defer cancel() + result := ToOpenAIResponsesRequest(ctx, req) + if result == nil { + t.Fatal("expected non-nil result") + } + + out := result.Input.OpenAIResponsesRequestInputArray + callID := *out[0].ResponsesToolMessage.CallID + outputCallID := *out[1].ResponsesToolMessage.CallID + + if callID != "search" { + t.Errorf("function_call id: got %q, want %q", callID, "search") + } + if len(callID) > 64 { + t.Errorf("function_call id exceeds OpenAI's 64-char limit: %d chars", len(callID)) + } + if outputCallID != callID { + t.Errorf("function_call_output id %q must match function_call id %q", outputCallID, callID) + } + + // The caller's history must be untouched so a later Gemini turn can recover the signature. + if *req.Input[0].ResponsesToolMessage.CallID != embeddedID { + t.Error("original function_call call_id was mutated") + } + if *req.Input[1].ResponsesToolMessage.CallID != embeddedID { + t.Error("original function_call_output call_id was mutated") + } +} diff --git a/core/providers/openai/responseslifecycle.go b/core/providers/openai/responseslifecycle.go new file mode 100644 index 00000000000..e5f1ca9a655 --- /dev/null +++ b/core/providers/openai/responseslifecycle.go @@ -0,0 +1,264 @@ +package openai + +import ( + "fmt" + "net/http" + "net/url" + "strconv" + + "github.com/valyala/fasthttp" + + providerUtils "github.com/maximhq/bifrost/core/providers/utils" + "github.com/maximhq/bifrost/core/schemas" +) + +func buildResponsesRetrieveQuery(req *schemas.BifrostResponsesRetrieveRequest) string { + if req == nil { + return "" + } + v := url.Values{} + for _, inc := range req.Include { + if inc != "" { + v.Add("include", inc) + } + } + if req.StartingAfter != nil { + v.Set("starting_after", strconv.Itoa(*req.StartingAfter)) + } + if req.IncludeObfuscation != nil { + v.Set("include_obfuscation", strconv.FormatBool(*req.IncludeObfuscation)) + } + return v.Encode() +} + +func buildResponsesInputItemsQuery(req *schemas.BifrostResponsesInputItemsRequest) string { + if req == nil { + return "" + } + v := url.Values{} + if req.After != "" { + v.Set("after", req.After) + } + for _, inc := range req.Include { + if inc != "" { + v.Add("include", inc) + } + } + if req.Limit != nil { + v.Set("limit", strconv.Itoa(*req.Limit)) + } + if req.Order != "" { + v.Set("order", req.Order) + } + return v.Encode() +} + +// executeResponsesLifecycleUnary performs a unary HTTP call for Responses lifecycle endpoints. +func (provider *OpenAIProvider) executeResponsesLifecycleUnary( + ctx *schemas.BifrostContext, + method string, + relativePath string, + requestType schemas.RequestType, + rawQuery string, + key schemas.Key, + body []byte, +) ([]byte, int64, map[string]string, *schemas.BifrostError) { + effectiveBody := body + fullURL := provider.buildRequestURL(ctx, relativePath, requestType) + if rawQuery != "" { + fullURL = fullURL + "?" + rawQuery + } + + req := fasthttp.AcquireRequest() + resp := fasthttp.AcquireResponse() + defer fasthttp.ReleaseRequest(req) + respOwned := true + defer func() { + if respOwned { + fasthttp.ReleaseResponse(resp) + } + }() + + // Lifecycle JSON is always consumed in-process (no transport streaming). Skip + // PrepareResponseStreaming so large-response threshold mode never leaves the body + // on a stream-only path that finalizeOpenAIResponse would treat as unsupported here. + activeClient := provider.client + providerUtils.SetExtraHeaders(ctx, req, provider.networkConfig.ExtraHeaders, nil) + + req.SetRequestURI(fullURL) + req.Header.SetMethod(method) + req.Header.SetContentType("application/json") + if len(body) > 0 { + req.SetBody(body) + } else if method == http.MethodPost { + effectiveBody = []byte("{}") + req.SetBody(effectiveBody) + } + + if key.Value.GetValue() != "" { + req.Header.Set("Authorization", "Bearer "+key.Value.GetValue()) + } + + sendBackRawRequest := providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest) + sendBackRawResponse := providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) + + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) + defer wait() + if bifrostErr != nil { + return nil, 0, nil, providerUtils.EnrichError(ctx, bifrostErr, effectiveBody, nil, sendBackRawRequest, sendBackRawResponse) + } + + providerResponseHeaders := providerUtils.ExtractProviderResponseHeaders(resp) + ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerResponseHeaders) + + if resp.StatusCode() != fasthttp.StatusOK { + providerUtils.MaterializeStreamErrorBody(ctx, resp) + return nil, 0, providerResponseHeaders, providerUtils.EnrichError(ctx, ParseOpenAIError(resp), effectiveBody, resp.Body(), sendBackRawRequest, sendBackRawResponse) + } + + bodyBytes, lpResult, finalErr := finalizeOpenAIResponse(ctx, resp, latency, provider.GetProviderKey(), provider.logger) + respOwned = false + if finalErr != nil { + return nil, 0, providerResponseHeaders, providerUtils.EnrichError(ctx, finalErr, effectiveBody, nil, sendBackRawRequest, sendBackRawResponse) + } + if lpResult != nil { + return nil, lpResult.Latency, providerResponseHeaders, providerUtils.NewBifrostOperationError( + schemas.ErrProviderResponseEmpty, + fmt.Errorf("responses lifecycle does not support large-response streaming mode"), + ) + } + + return bodyBytes, latency.Milliseconds(), providerResponseHeaders, nil +} + +// ResponsesRetrieve implements schemas.ResponsesLifecycleProvider. +func (provider *OpenAIProvider) ResponsesRetrieve(ctx *schemas.BifrostContext, key schemas.Key, req *schemas.BifrostResponsesRetrieveRequest) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) { + if err := providerUtils.CheckOperationAllowed(schemas.OpenAI, provider.customProviderConfig, schemas.ResponsesRetrieveRequest); err != nil { + return nil, err + } + if req == nil || req.ResponseID == "" { + return nil, providerUtils.NewBifrostOperationError(schemas.ErrRequestBodyConversion, fmt.Errorf("response_id is required")) + } + + path := "/v1/responses/" + url.PathEscape(req.ResponseID) + bodyBytes, latencyMs, headers, bifrostErr := provider.executeResponsesLifecycleUnary( + ctx, http.MethodGet, path, schemas.ResponsesRetrieveRequest, buildResponsesRetrieveQuery(req), key, nil) + if bifrostErr != nil { + return nil, bifrostErr + } + + response := &schemas.BifrostResponsesResponse{} + sendBackRawRequest := providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest) + sendBackRawResponse := providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) + _, rawResponse, err := providerUtils.HandleProviderResponse(bodyBytes, response, nil, sendBackRawRequest, sendBackRawResponse) + if err != nil { + return nil, providerUtils.EnrichError(ctx, err, nil, bodyBytes, sendBackRawRequest, sendBackRawResponse) + } + response.ExtraFields.Latency = latencyMs + response.ExtraFields.ProviderResponseHeaders = headers + response.ExtraFields.Provider = provider.GetProviderKey() + if sendBackRawResponse { + response.ExtraFields.RawResponse = rawResponse + } + return response, nil +} + +// ResponsesDelete implements schemas.ResponsesLifecycleProvider. +func (provider *OpenAIProvider) ResponsesDelete(ctx *schemas.BifrostContext, key schemas.Key, req *schemas.BifrostResponsesDeleteRequest) (*schemas.BifrostResponsesDeleteResponse, *schemas.BifrostError) { + if err := providerUtils.CheckOperationAllowed(schemas.OpenAI, provider.customProviderConfig, schemas.ResponsesDeleteRequest); err != nil { + return nil, err + } + if req == nil || req.ResponseID == "" { + return nil, providerUtils.NewBifrostOperationError(schemas.ErrRequestBodyConversion, fmt.Errorf("response_id is required")) + } + + path := "/v1/responses/" + url.PathEscape(req.ResponseID) + bodyBytes, latencyMs, headers, bifrostErr := provider.executeResponsesLifecycleUnary( + ctx, http.MethodDelete, path, schemas.ResponsesDeleteRequest, "", key, nil) + if bifrostErr != nil { + return nil, bifrostErr + } + + response := &schemas.BifrostResponsesDeleteResponse{} + sendBackRawRequest := providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest) + sendBackRawResponse := providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) + _, rawResponse, err := providerUtils.HandleProviderResponse(bodyBytes, response, nil, sendBackRawRequest, sendBackRawResponse) + if err != nil { + return nil, providerUtils.EnrichError(ctx, err, nil, bodyBytes, sendBackRawRequest, sendBackRawResponse) + } + response.ExtraFields.Latency = latencyMs + response.ExtraFields.ProviderResponseHeaders = headers + response.ExtraFields.Provider = provider.GetProviderKey() + if sendBackRawResponse { + response.ExtraFields.RawResponse = rawResponse + } + return response, nil +} + +// ResponsesCancel implements schemas.ResponsesLifecycleProvider. +func (provider *OpenAIProvider) ResponsesCancel(ctx *schemas.BifrostContext, key schemas.Key, req *schemas.BifrostResponsesCancelRequest) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) { + if err := providerUtils.CheckOperationAllowed(schemas.OpenAI, provider.customProviderConfig, schemas.ResponsesCancelRequest); err != nil { + return nil, err + } + if req == nil || req.ResponseID == "" { + return nil, providerUtils.NewBifrostOperationError(schemas.ErrRequestBodyConversion, fmt.Errorf("response_id is required")) + } + + path := "/v1/responses/" + url.PathEscape(req.ResponseID) + "/cancel" + bodyBytes, latencyMs, headers, bifrostErr := provider.executeResponsesLifecycleUnary( + ctx, http.MethodPost, path, schemas.ResponsesCancelRequest, "", key, nil) + if bifrostErr != nil { + return nil, bifrostErr + } + + response := &schemas.BifrostResponsesResponse{} + sendBackRawRequest := providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest) + sendBackRawResponse := providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) + cancelBody := []byte("{}") + rawRequest, rawResponse, err := providerUtils.HandleProviderResponse(bodyBytes, response, cancelBody, sendBackRawRequest, sendBackRawResponse) + if err != nil { + return nil, providerUtils.EnrichError(ctx, err, cancelBody, bodyBytes, sendBackRawRequest, sendBackRawResponse) + } + response.ExtraFields.Latency = latencyMs + response.ExtraFields.ProviderResponseHeaders = headers + response.ExtraFields.Provider = provider.GetProviderKey() + if sendBackRawRequest { + response.ExtraFields.RawRequest = rawRequest + } + if sendBackRawResponse { + response.ExtraFields.RawResponse = rawResponse + } + return response, nil +} + +// ResponsesInputItems implements schemas.ResponsesLifecycleProvider. +func (provider *OpenAIProvider) ResponsesInputItems(ctx *schemas.BifrostContext, key schemas.Key, req *schemas.BifrostResponsesInputItemsRequest) (*schemas.BifrostResponsesInputItemsResponse, *schemas.BifrostError) { + if err := providerUtils.CheckOperationAllowed(schemas.OpenAI, provider.customProviderConfig, schemas.ResponsesInputItemsRequest); err != nil { + return nil, err + } + if req == nil || req.ResponseID == "" { + return nil, providerUtils.NewBifrostOperationError(schemas.ErrRequestBodyConversion, fmt.Errorf("response_id is required")) + } + + path := "/v1/responses/" + url.PathEscape(req.ResponseID) + "/input_items" + bodyBytes, latencyMs, headers, bifrostErr := provider.executeResponsesLifecycleUnary( + ctx, http.MethodGet, path, schemas.ResponsesInputItemsRequest, buildResponsesInputItemsQuery(req), key, nil) + if bifrostErr != nil { + return nil, bifrostErr + } + + response := &schemas.BifrostResponsesInputItemsResponse{} + sendBackRawRequest := providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest) + sendBackRawResponse := providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) + _, rawResponse, err := providerUtils.HandleProviderResponse(bodyBytes, response, nil, sendBackRawRequest, sendBackRawResponse) + if err != nil { + return nil, providerUtils.EnrichError(ctx, err, nil, bodyBytes, sendBackRawRequest, sendBackRawResponse) + } + response.ExtraFields.Latency = latencyMs + response.ExtraFields.ProviderResponseHeaders = headers + response.ExtraFields.Provider = provider.GetProviderKey() + if sendBackRawResponse { + response.ExtraFields.RawResponse = rawResponse + } + return response, nil +} diff --git a/core/providers/openai/tool_search_roundtrip_test.go b/core/providers/openai/tool_search_roundtrip_test.go new file mode 100644 index 00000000000..4bb43eaafee --- /dev/null +++ b/core/providers/openai/tool_search_roundtrip_test.go @@ -0,0 +1,53 @@ +package openai + +import ( + "testing" + + "github.com/tidwall/gjson" +) + +// TestResponsesInputRoundTripsToolSearchItems reproduces the codex deferral +// follow-up request that previously made Bifrost reject the whole input array +// with "openai responses request input is neither a string nor an array of +// responses messages" (tool_search_call.arguments is an OBJECT, not the string +// function_call uses), and verifies the items now round-trip unchanged. +func TestResponsesInputRoundTripsToolSearchItems(t *testing.T) { + input := []byte(`[ + {"role":"user","content":[{"type":"input_text","text":"Read REQ-1"}]}, + {"type":"tool_search_call","call_id":"s1","execution":"client","arguments":{"query":"ticket","limit":1}}, + {"type":"tool_search_output","call_id":"s1","status":"completed","execution":"client","tools":[{"type":"function","name":"get_request","description":"x","defer_loading":true,"parameters":{"type":"object","properties":{"id":{"type":"string"}},"required":["id"],"additionalProperties":false}}]} + ]`) + + var in OpenAIResponsesRequestInput + if err := in.UnmarshalJSON(input); err != nil { + t.Fatalf("unmarshal failed (the original bug): %v", err) + } + if got := len(in.OpenAIResponsesRequestInputArray); got != 3 { + t.Fatalf("expected 3 input items, got %d", got) + } + + out, err := in.MarshalJSON() + if err != nil { + t.Fatalf("marshal failed: %v", err) + } + + // tool_search_call.arguments must stay a JSON OBJECT (OpenAI 400s a string). + if !gjson.GetBytes(out, "1.arguments").IsObject() { + t.Fatalf("tool_search_call.arguments is not an object: %s", gjson.GetBytes(out, "1.arguments").Raw) + } + if got := gjson.GetBytes(out, "1.arguments.query").String(); got != "ticket" { + t.Fatalf("tool_search_call.arguments.query lost: %q", got) + } + // tool_search_output.tools[0].type must survive (OpenAI requires it). + if got := gjson.GetBytes(out, "2.tools.0.type").String(); got != "function" { + t.Fatalf("tool_search_output.tools[0].type lost: %q (full: %s)", got, gjson.GetBytes(out, "2.tools").Raw) + } + if got := gjson.GetBytes(out, "2.tools.0.name").String(); got != "get_request" { + t.Fatalf("tool_search_output.tools[0].name lost: %q", got) + } + // The ordinary user message must still parse/serialize normally. + if got := gjson.GetBytes(out, "0.role").String(); got != "user" { + t.Fatalf("plain user message broke: %q", got) + } + t.Logf("round-trip OK:\n%s", out) +} diff --git a/core/providers/openai/transcription.go b/core/providers/openai/transcription.go index cbfb1307145..e0b1081797d 100644 --- a/core/providers/openai/transcription.go +++ b/core/providers/openai/transcription.go @@ -3,6 +3,7 @@ package openai import ( "fmt" "mime/multipart" + "sort" "github.com/maximhq/bifrost/core/providers/utils" "github.com/maximhq/bifrost/core/schemas" @@ -40,6 +41,7 @@ func ToOpenAITranscriptionRequest(bifrostReq *schemas.BifrostTranscriptionReques if params != nil { openaiReq.TranscriptionParameters = *params + openaiReq.ExtraParams = params.ExtraParams } return openaiReq @@ -95,6 +97,35 @@ func ParseTranscriptionFormDataBodyFromRequest(writer *multipart.Writer, openaiR } } + // Forward provider-specific passthrough params (e.g. chunking_strategy, required by + // OpenAI diarization models). String values are written verbatim; object values are + // encoded as JSON since multipart form fields are strings. Keys are sorted so the + // emitted form is deterministic. + if len(openaiReq.ExtraParams) > 0 { + extraKeys := make([]string, 0, len(openaiReq.ExtraParams)) + for key := range openaiReq.ExtraParams { + extraKeys = append(extraKeys, key) + } + sort.Strings(extraKeys) + for _, key := range extraKeys { + value := openaiReq.ExtraParams[key] + var fieldValue string + switch v := value.(type) { + case string: + fieldValue = v + default: + encoded, err := schemas.MarshalSorted(v) + if err != nil { + return utils.NewBifrostOperationError(fmt.Sprintf("failed to encode %s field", key), err) + } + fieldValue = string(encoded) + } + if err := writer.WriteField(key, fieldValue); err != nil { + return utils.NewBifrostOperationError(fmt.Sprintf("failed to write %s field", key), err) + } + } + } + // Add file field last so large multipart uploads don't block model discovery upstream. filename := openaiReq.Filename if filename == "" { diff --git a/core/providers/openai/transcription_test.go b/core/providers/openai/transcription_test.go index ddbb33b888d..1def4c5c4be 100644 --- a/core/providers/openai/transcription_test.go +++ b/core/providers/openai/transcription_test.go @@ -2,6 +2,7 @@ package openai import ( "bytes" + "encoding/json" "io" "mime" "mime/multipart" @@ -77,3 +78,95 @@ func TestParseTranscriptionFormDataBodyFromRequest_OrdersMetadataBeforeFile(t *t t.Fatalf("expected model part first, got order %v", order) } } + +// multipartFieldValue returns the value of the first multipart part with the +// given name, or "" if absent. +func multipartFieldValue(t *testing.T, contentType string, body []byte, name string) string { + t.Helper() + _, params, err := mime.ParseMediaType(contentType) + if err != nil { + t.Fatalf("ParseMediaType(%q): %v", contentType, err) + } + boundary := params["boundary"] + reader := multipart.NewReader(bytes.NewReader(body), boundary) + for { + part, err := reader.NextPart() + if err == io.EOF { + break + } + if err != nil { + t.Fatalf("NextPart(): %v", err) + } + if part.FormName() == name { + data, _ := io.ReadAll(part) + _ = part.Close() + return string(data) + } + _, _ = io.Copy(io.Discard, part) + _ = part.Close() + } + return "" +} + +func TestParseTranscriptionFormDataBodyFromRequest_ChunkingStrategyString(t *testing.T) { + var body bytes.Buffer + writer := multipart.NewWriter(&body) + req := &OpenAITranscriptionRequest{ + Model: "gpt-4o-transcribe-diarize", + File: []byte("audio-bytes"), + Filename: "sample.mp3", + TranscriptionParameters: schemas.TranscriptionParameters{ + ExtraParams: map[string]interface{}{ + "chunking_strategy": "auto", + }, + }, + } + + if bifrostErr := ParseTranscriptionFormDataBodyFromRequest(writer, req, schemas.OpenAI); bifrostErr != nil { + t.Fatalf("unexpected bifrost error: %v", bifrostErr.Error.Message) + } + + contentType := writer.FormDataContentType() + if got := multipartFieldValue(t, contentType, body.Bytes(), "chunking_strategy"); got != "auto" { + t.Fatalf("expected chunking_strategy=auto written verbatim, got %q", got) + } + + order := multipartPartOrder(t, contentType, body.Bytes()) + if order[len(order)-1] != "file" { + t.Fatalf("expected file part last, got order %v", order) + } +} + +func TestParseTranscriptionFormDataBodyFromRequest_ChunkingStrategyObject(t *testing.T) { + var body bytes.Buffer + writer := multipart.NewWriter(&body) + req := &OpenAITranscriptionRequest{ + Model: "gpt-4o-transcribe-diarize", + File: []byte("audio-bytes"), + Filename: "sample.mp3", + TranscriptionParameters: schemas.TranscriptionParameters{ + ExtraParams: map[string]interface{}{ + "chunking_strategy": map[string]interface{}{ + "type": "server_vad", + "threshold": 0.5, + }, + }, + }, + } + + if bifrostErr := ParseTranscriptionFormDataBodyFromRequest(writer, req, schemas.OpenAI); bifrostErr != nil { + t.Fatalf("unexpected bifrost error: %v", bifrostErr.Error.Message) + } + + got := multipartFieldValue(t, writer.FormDataContentType(), body.Bytes(), "chunking_strategy") + if got == "" { + t.Fatal("expected chunking_strategy object to be written as a form field") + } + var decoded map[string]interface{} + if err := json.Unmarshal([]byte(got), &decoded); err != nil { + t.Fatalf("expected chunking_strategy to be valid JSON, got %q: %v", got, err) + } + if decoded["type"] != "server_vad" { + t.Fatalf("expected type=server_vad, got %v", decoded["type"]) + } +} diff --git a/core/providers/openai/utils.go b/core/providers/openai/utils.go index 881324980a4..1da19c7afca 100644 --- a/core/providers/openai/utils.go +++ b/core/providers/openai/utils.go @@ -3,6 +3,7 @@ package openai import ( "strings" + "github.com/maximhq/bifrost/core/providers/utils" "github.com/maximhq/bifrost/core/schemas" ) @@ -40,12 +41,46 @@ func ConvertBifrostMessagesToOpenAIMessages(messages []schemas.ChatMessage) []Op Content: message.Content, ChatToolMessage: message.ChatToolMessage, } + // Strip provider reasoning signatures (e.g. Gemini thoughtSignatures embedded in + // call_id as "_ts_") from the tool result's tool_call_id, but only when it + // exceeds OpenAI's limit — shorter IDs are left intact so distinct upstream IDs are + // preserved. Clone first — ChatToolMessage is shared with the caller's input. + if message.ChatToolMessage != nil && message.ChatToolMessage.ToolCallID != nil && + len(*message.ChatToolMessage.ToolCallID) > MaxToolCallIDLength { + if stripped := utils.StripThoughtSignature(*message.ChatToolMessage.ToolCallID); stripped != *message.ChatToolMessage.ToolCallID { + toolMsgCopy := *message.ChatToolMessage + toolMsgCopy.ToolCallID = &stripped + openaiMessages[i].ChatToolMessage = &toolMsgCopy + } + } if message.ChatAssistantMessage != nil { + // Strip the same embedded signature from over-long assistant tool call IDs. Clone the + // slice only when a strip is actually needed so the caller's input is never mutated. + toolCalls := message.ChatAssistantMessage.ToolCalls + needsStrip := false + for j := range toolCalls { + if toolCalls[j].ID != nil && len(*toolCalls[j].ID) > MaxToolCallIDLength && + strings.Contains(*toolCalls[j].ID, utils.ThoughtSignatureSeparator) { + needsStrip = true + break + } + } + if needsStrip { + cloned := make([]schemas.ChatAssistantMessageToolCall, len(toolCalls)) + copy(cloned, toolCalls) + for j := range cloned { + if cloned[j].ID != nil && len(*cloned[j].ID) > MaxToolCallIDLength { + stripped := utils.StripThoughtSignature(*cloned[j].ID) + cloned[j].ID = &stripped + } + } + toolCalls = cloned + } openaiMessages[i].OpenAIChatAssistantMessage = &OpenAIChatAssistantMessage{ Refusal: message.ChatAssistantMessage.Refusal, Reasoning: message.ChatAssistantMessage.Reasoning, Annotations: message.ChatAssistantMessage.Annotations, - ToolCalls: message.ChatAssistantMessage.ToolCalls, + ToolCalls: toolCalls, } } } @@ -144,6 +179,9 @@ func supportsMaxReasoningEffort(model string) bool { // MaxUserFieldLength for OpenAI enforces a 64 character maximum on the user field const MaxUserFieldLength = 64 +// MaxToolCallIDLength is OpenAI's 64 character maximum on tool call IDs (call_id / input[].id). +const MaxToolCallIDLength = 64 + // SanitizeUserField returns nil if user exceeds MaxUserFieldLength, otherwise returns the original value func SanitizeUserField(user *string) *string { if user != nil && len(*user) > MaxUserFieldLength { diff --git a/core/providers/opencode/opencode.go b/core/providers/opencode/opencode.go index 2bddf26bb0b..bbca24516ed 100644 --- a/core/providers/opencode/opencode.go +++ b/core/providers/opencode/opencode.go @@ -115,29 +115,26 @@ func (p *opencodeProvider) ChatCompletion(ctx *schemas.BifrostContext, key schem p.client, p.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - key, + openai.BearerAuthHeader(key), p.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, p.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, p.sendBackRawResponse), p.GetProviderKey(), nil, parseOpencodeError, + nil, p.logger, ) } // ChatCompletionStream performs a streaming chat completion request to the Opencode API. func (p *opencodeProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - if v := key.Value.GetValue(); v != "" { - authHeader = map[string]string{"Authorization": "Bearer " + v} - } return openai.HandleOpenAIChatCompletionStreaming( ctx, p.streamingClient, p.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - authHeader, + openai.BearerAuthHeader(key), p.networkConfig.ExtraHeaders, p.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, p.sendBackRawRequest), @@ -149,6 +146,7 @@ func (p *opencodeProvider) ChatCompletionStream(ctx *schemas.BifrostContext, pos parseOpencodeError, nil, nil, + nil, p.logger, postHookSpanFinalizer, ) diff --git a/core/providers/openrouter/openrouter.go b/core/providers/openrouter/openrouter.go index 516f52477e4..74ebd9ca03a 100644 --- a/core/providers/openrouter/openrouter.go +++ b/core/providers/openrouter/openrouter.go @@ -89,7 +89,7 @@ func (provider *OpenRouterProvider) validateKey(ctx *schemas.BifrostContext, key providerUtils.SetExtraHeaders(ctx, req, provider.networkConfig.ExtraHeaders, nil) // Make request - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { return bifrostErr @@ -98,12 +98,12 @@ func (provider *OpenRouterProvider) validateKey(ctx *schemas.BifrostContext, key // Check for auth errors (401, 403) statusCode := resp.StatusCode() if statusCode == fasthttp.StatusUnauthorized || statusCode == fasthttp.StatusForbidden { - return openai.ParseOpenAIError(resp) + return providerUtils.SetErrorLatency(openai.ParseOpenAIError(resp), latency) } // Any 4xx/5xx error indicates the key might be invalid if statusCode >= 400 { - return openai.ParseOpenAIError(resp) + return providerUtils.SetErrorLatency(openai.ParseOpenAIError(resp), latency) } return nil @@ -159,7 +159,7 @@ func (provider *OpenRouterProvider) listModelsByKey(ctx *schemas.BifrostContext, // Continue with empty response; allowed models will be backfilled below. modelsFetched = false } else { - bifrostErr := openai.ParseOpenAIError(resp) + bifrostErr := providerUtils.SetErrorLatency(openai.ParseOpenAIError(resp), latency) return nil, bifrostErr } } @@ -269,7 +269,7 @@ func (provider *OpenRouterProvider) TextCompletion(ctx *schemas.BifrostContext, provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -284,17 +284,12 @@ func (provider *OpenRouterProvider) TextCompletion(ctx *schemas.BifrostContext, // It formats the request, sends it to OpenRouter, and processes the response. // Returns a channel of BifrostStreamChunk objects or an error if the request fails. func (provider *OpenRouterProvider) TextCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostTextCompletionRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - keyValue := key.Value.GetValue() - if keyValue != "" { - authHeader = map[string]string{"Authorization": "Bearer " + keyValue} - } return openai.HandleOpenAITextCompletionStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+"/v1/completions", request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -316,13 +311,14 @@ func (provider *OpenRouterProvider) ChatCompletion(ctx *schemas.BifrostContext, provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), nil, nil, + nil, provider.logger, ) } @@ -332,18 +328,12 @@ func (provider *OpenRouterProvider) ChatCompletion(ctx *schemas.BifrostContext, // Uses OpenRouter's OpenAI-compatible streaming format. // Returns a channel containing BifrostStreamChunk objects representing the stream or an error if the request fails. func (provider *OpenRouterProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - keyValue := key.Value.GetValue() - if keyValue != "" { - authHeader = map[string]string{"Authorization": "Bearer " + keyValue} - } - // Use shared OpenAI-compatible streaming logic return openai.HandleOpenAIChatCompletionStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -355,6 +345,7 @@ func (provider *OpenRouterProvider) ChatCompletionStream(ctx *schemas.BifrostCon nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -367,30 +358,26 @@ func (provider *OpenRouterProvider) Responses(ctx *schemas.BifrostContext, key s provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/responses"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), nil, nil, + nil, provider.logger, ) } // ResponsesStream performs a streaming responses request to the OpenRouter API. func (provider *OpenRouterProvider) ResponsesStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostResponsesRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - keyValue := key.Value.GetValue() - if keyValue != "" { - authHeader = map[string]string{"Authorization": "Bearer " + keyValue} - } return openai.HandleOpenAIResponsesStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/responses"), request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -401,6 +388,7 @@ func (provider *OpenRouterProvider) ResponsesStream(ctx *schemas.BifrostContext, nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -413,7 +401,7 @@ func (provider *OpenRouterProvider) Embedding(ctx *schemas.BifrostContext, key s provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/embeddings"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), diff --git a/core/providers/parasail/parasail.go b/core/providers/parasail/parasail.go index 1a5df4a7e4a..d7275a89bbc 100644 --- a/core/providers/parasail/parasail.go +++ b/core/providers/parasail/parasail.go @@ -100,13 +100,14 @@ func (provider *ParasailProvider) ChatCompletion(ctx *schemas.BifrostContext, ke provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), nil, nil, + nil, provider.logger, ) } @@ -116,17 +117,12 @@ func (provider *ParasailProvider) ChatCompletion(ctx *schemas.BifrostContext, ke // Uses Parasail's OpenAI-compatible streaming format. // Returns a channel containing BifrostStreamChunk objects representing the stream or an error if the request fails. func (provider *ParasailProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } - // Use shared OpenAI-compatible streaming logic return openai.HandleOpenAIChatCompletionStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+"/v1/chat/completions", request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -138,6 +134,7 @@ func (provider *ParasailProvider) ChatCompletionStream(ctx *schemas.BifrostConte nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) diff --git a/core/providers/perplexity/perplexity.go b/core/providers/perplexity/perplexity.go index 3a87fd6cab2..aa7d606d88d 100644 --- a/core/providers/perplexity/perplexity.go +++ b/core/providers/perplexity/perplexity.go @@ -105,7 +105,7 @@ func (provider *PerplexityProvider) completeRequest(ctx *schemas.BifrostContext, // Handle error response if resp.StatusCode() != fasthttp.StatusOK { provider.logger.Debug(fmt.Sprintf("error from %s provider: %s", provider.GetProviderKey(), string(resp.Body()))) - return nil, latency, providerResponseHeaders, openai.ParseOpenAIError(resp) + return nil, latency, providerResponseHeaders, providerUtils.SetErrorLatency(openai.ParseOpenAIError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -121,8 +121,19 @@ func (provider *PerplexityProvider) completeRequest(ctx *schemas.BifrostContext, } // ListModels performs a list models request to Perplexity's API. +// Perplexity's /v1/models endpoint is OpenAI-compatible, so the OpenAI handler is reused. func (provider *PerplexityProvider) ListModels(ctx *schemas.BifrostContext, keys []schemas.Key, request *schemas.BifrostListModelsRequest) (*schemas.BifrostListModelsResponse, *schemas.BifrostError) { - return nil, providerUtils.NewUnsupportedOperationError(schemas.ListModelsRequest, provider.GetProviderKey()) + return openai.HandleOpenAIListModelsRequest( + ctx, + provider.client, + request, + provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/models"), + keys, + provider.networkConfig.ExtraHeaders, + provider.GetProviderKey(), + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + ) } // TextCompletion is not supported by the Perplexity provider. @@ -152,13 +163,13 @@ func (provider *PerplexityProvider) ChatCompletion(ctx *schemas.BifrostContext, responseBody, latency, providerResponseHeaders, err := provider.completeRequest(ctx, jsonBody, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/chat/completions"), key.Value.GetValue(), request.Model) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } var response PerplexityChatResponse rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, &response, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse := response.ToBifrostChatResponse(request.Model) @@ -185,22 +196,17 @@ func (provider *PerplexityProvider) ChatCompletion(ctx *schemas.BifrostContext, // Uses Perplexity's OpenAI-compatible streaming format. // Returns a channel containing BifrostStreamChunk objects representing the stream or an error if the request fails. func (provider *PerplexityProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } customRequestConverter := func(request *schemas.BifrostChatRequest) (providerUtils.RequestBodyWithExtraParams, error) { reqBody := ToPerplexityChatCompletionRequest(request) reqBody.Stream = schemas.Ptr(true) return reqBody, nil } - // Use shared OpenAI-compatible streaming logic return openai.HandleOpenAIChatCompletionStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+"/chat/completions", request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -212,32 +218,75 @@ func (provider *PerplexityProvider) ChatCompletionStream(ctx *schemas.BifrostCon nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) } // Responses performs a responses request to the Perplexity API. +// Models available on Perplexity's /v1/responses endpoint are routed there via the +// OpenAI-compatible handler; sonar-* models not supported on responses fall back to /chat/completions. func (provider *PerplexityProvider) Responses(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostResponsesRequest) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) { - chatResponse, err := provider.ChatCompletion(ctx, key, request.ToChatRequest()) - if err != nil { - return nil, err + if !isPerplexityResponsesSupported(schemas.ResolveCanonicalModel(ctx, request.Model)) { + chatResponse, err := provider.ChatCompletion(ctx, key, request.ToChatRequest()) + if err != nil { + return nil, err + } + return chatResponse.ToBifrostResponsesResponse(), nil } - response := chatResponse.ToBifrostResponsesResponse() - - return response, nil + return openai.HandleOpenAIResponsesRequest( + ctx, + provider.client, + provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/responses"), + request, + openai.BearerAuthHeader(key), + provider.networkConfig.ExtraHeaders, + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + schemas.Perplexity, + nil, + nil, + nil, + provider.logger, + ) } // ResponsesStream performs a streaming responses request to the Perplexity API. +// Models available on Perplexity's /v1/responses endpoint are streamed from there via the +// OpenAI-compatible handler; sonar-* models not supported on responses fall back to /chat/completions. func (provider *PerplexityProvider) ResponsesStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostResponsesRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - ctx.SetValue(schemas.BifrostContextKeyIsResponsesToChatCompletionFallback, true) - return provider.ChatCompletionStream( + if !isPerplexityResponsesSupported(schemas.ResolveCanonicalModel(ctx, request.Model)) { + ctx.SetValue(schemas.BifrostContextKeyIsResponsesToChatCompletionFallback, true) + return provider.ChatCompletionStream( + ctx, + postHookRunner, + postHookSpanFinalizer, + key, + request.ToChatRequest(), + ) + } + + return openai.HandleOpenAIResponsesStreaming( ctx, + provider.streamingClient, + provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/responses"), + request, + openai.BearerAuthHeader(key), + provider.networkConfig.ExtraHeaders, + provider.networkConfig.StreamIdleTimeoutInSeconds, + providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), + providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), + schemas.Perplexity, postHookRunner, + nil, + nil, + nil, + nil, + nil, + provider.logger, postHookSpanFinalizer, - key, - request.ToChatRequest(), ) } diff --git a/core/providers/perplexity/responses.go b/core/providers/perplexity/responses.go index fe229e99f59..5f21841eaa3 100644 --- a/core/providers/perplexity/responses.go +++ b/core/providers/perplexity/responses.go @@ -1,9 +1,19 @@ package perplexity import ( + "strings" + "github.com/maximhq/bifrost/core/schemas" ) +// isPerplexityResponsesSupported reports whether the model should use /v1/responses vs /chat/completions. +// Denylist by design: /chat/completions serves only the Sonar family, so sonar-* variants go to chat and +// every other model (base sonar + non-Sonar) goes to responses. This keeps newly-shipped non-Sonar models +// working without a code change; they'd fail on chat, which doesn't serve them. +func isPerplexityResponsesSupported(model string) bool { + return !strings.HasPrefix(strings.TrimPrefix(model, "perplexity/"), "sonar-") +} + // ToPerplexityResponsesRequest converts a BifrostResponsesRequest to PerplexityChatRequest func ToPerplexityResponsesRequest(bifrostReq *schemas.BifrostResponsesRequest) *PerplexityChatRequest { if bifrostReq == nil { diff --git a/core/providers/replicate/replicate.go b/core/providers/replicate/replicate.go index 151ed4d3b59..6bcd8b5b216 100644 --- a/core/providers/replicate/replicate.go +++ b/core/providers/replicate/replicate.go @@ -163,7 +163,7 @@ func createPrediction( // Handle error response if resp.StatusCode() != fasthttp.StatusOK && resp.StatusCode() != fasthttp.StatusCreated { logger.Debug(fmt.Sprintf("error from replicate provider: %s", string(resp.Body()))) - return nil, nil, latency, providerResponseHeaders, parseReplicateError(resp.Body(), resp.StatusCode()) + return nil, nil, latency, providerResponseHeaders, providerUtils.SetErrorLatency(parseReplicateError(resp.Body(), resp.StatusCode()), latency) } // Parse response @@ -206,7 +206,7 @@ func getPrediction( } // Make request - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, client, req, resp) defer wait() if bifrostErr != nil { return nil, nil, nil, bifrostErr @@ -218,7 +218,7 @@ func getPrediction( // Handle error response if resp.StatusCode() != fasthttp.StatusOK { logger.Debug(fmt.Sprintf("error from replicate provider: %s", string(resp.Body()))) - return nil, nil, providerResponseHeaders, parseReplicateError(resp.Body(), resp.StatusCode()) + return nil, nil, providerResponseHeaders, providerUtils.SetErrorLatency(parseReplicateError(resp.Body(), resp.StatusCode()), latency) } // Parse response @@ -333,7 +333,7 @@ func (provider *ReplicateProvider) listDeploymentsByKey(ctx *schemas.BifrostCont providerUtils.SetExtraHeaders(ctx, req, extraHeaders, nil) // Make request - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, client, req, resp) // Release resources wait() @@ -346,7 +346,7 @@ func (provider *ReplicateProvider) listDeploymentsByKey(ctx *schemas.BifrostCont // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - errorResponse := parseReplicateError(resp.Body(), resp.StatusCode()) + errorResponse := providerUtils.SetErrorLatency(parseReplicateError(resp.Body(), resp.StatusCode()), latency) fasthttp.ReleaseResponse(resp) return nil, errorResponse } @@ -467,7 +467,7 @@ func (provider *ReplicateProvider) TextCompletion(ctx *schemas.BifrostContext, k provider.sendBackRawResponse, ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // if not sync, poll until done @@ -482,13 +482,13 @@ func (provider *ReplicateProvider) TextCompletion(ctx *schemas.BifrostContext, k providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } } // Check for terminal error status (failed/canceled) after sync mode or polling if err := checkForErrorStatus(prediction); err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if providerResponseHeaders != nil { @@ -547,7 +547,7 @@ func (provider *ReplicateProvider) TextCompletionStream(ctx *schemas.BifrostCont startTime := time.Now() // Create prediction - prediction, _, _, _, err := createPrediction( + prediction, _, latency, _, err := createPrediction( ctx, provider.client, jsonData, @@ -560,7 +560,7 @@ func (provider *ReplicateProvider) TextCompletionStream(ctx *schemas.BifrostCont provider.sendBackRawResponse, ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Verify stream URL is available @@ -568,7 +568,7 @@ func (provider *ReplicateProvider) TextCompletionStream(ctx *schemas.BifrostCont bifrostErr := providerUtils.NewBifrostOperationError( "stream URL not available in prediction response", fmt.Errorf("prediction response missing stream URL")) - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } streamURL := *prediction.URLs.Stream @@ -576,12 +576,14 @@ func (provider *ReplicateProvider) TextCompletionStream(ctx *schemas.BifrostCont // Connect to stream URL _, resp, bifrostErr := listenToReplicateStreamURL(ctx, provider.streamingClient, streamURL, key) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Store provider response headers in context for transport layer ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerUtils.ExtractProviderResponseHeaders(resp)) + providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) + // Large payload streaming passthrough — pipe raw upstream SSE to client if providerUtils.SetupStreamingPassthrough(ctx, resp) { responseChan := make(chan *schemas.BifrostStreamChunk) @@ -592,8 +594,6 @@ func (provider *ReplicateProvider) TextCompletionStream(ctx *schemas.BifrostCont // Create response channel responseChan := make(chan *schemas.BifrostStreamChunk, schemas.DefaultStreamBufferSize) - providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) - // Start streaming in a goroutine go func() { defer providerUtils.EnsureStreamFinalizerCalled(ctx, postHookSpanFinalizer) @@ -645,7 +645,7 @@ func (provider *ReplicateProvider) TextCompletionStream(ctx *schemas.BifrostCont if readErr != io.EOF { ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) provider.logger.Warn("Error reading stream: %v", readErr) - enrichedErr := providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, readErr), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + enrichedErr := providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, readErr), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, enrichedErr, responseChan, provider.logger, postHookSpanFinalizer) } break @@ -713,7 +713,7 @@ func (provider *ReplicateProvider) TextCompletionStream(ctx *schemas.BifrostCont "prediction was canceled", fmt.Errorf("stream ended: prediction canceled")) ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - enrichedErr := providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + enrichedErr := providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, enrichedErr, responseChan, provider.logger, postHookSpanFinalizer) // Explicitly close the body stream to terminate connection to Replicate resp.CloseBodyStream() @@ -728,7 +728,7 @@ func (provider *ReplicateProvider) TextCompletionStream(ctx *schemas.BifrostCont errorMsg, fmt.Errorf("stream ended with error")) ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - enrichedErr := providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + enrichedErr := providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, enrichedErr, responseChan, provider.logger, postHookSpanFinalizer) // Explicitly close the body stream to terminate connection to Replicate resp.CloseBodyStream() @@ -807,7 +807,7 @@ func (provider *ReplicateProvider) ChatCompletion(ctx *schemas.BifrostContext, k provider.sendBackRawResponse, ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // if not sync, poll until done @@ -822,13 +822,13 @@ func (provider *ReplicateProvider) ChatCompletion(ctx *schemas.BifrostContext, k providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } } // Check for terminal error status (failed/canceled) after sync mode or polling if err := checkForErrorStatus(prediction); err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if providerResponseHeaders != nil { @@ -887,7 +887,7 @@ func (provider *ReplicateProvider) ChatCompletionStream(ctx *schemas.BifrostCont startTime := time.Now() // Create prediction - prediction, _, _, _, err := createPrediction( + prediction, _, latency, _, err := createPrediction( ctx, provider.client, jsonData, @@ -900,7 +900,7 @@ func (provider *ReplicateProvider) ChatCompletionStream(ctx *schemas.BifrostCont provider.sendBackRawResponse, ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Verify stream URL is available @@ -908,7 +908,7 @@ func (provider *ReplicateProvider) ChatCompletionStream(ctx *schemas.BifrostCont bifrostErr := providerUtils.NewBifrostOperationError( "stream URL not available in prediction response", fmt.Errorf("prediction response missing stream URL")) - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } streamURL := *prediction.URLs.Stream @@ -916,12 +916,14 @@ func (provider *ReplicateProvider) ChatCompletionStream(ctx *schemas.BifrostCont // Connect to stream URL _, resp, bifrostErr := listenToReplicateStreamURL(ctx, provider.streamingClient, streamURL, key) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Store provider response headers in context for transport layer ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerUtils.ExtractProviderResponseHeaders(resp)) + providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) + // Large payload streaming passthrough — pipe raw upstream SSE to client if providerUtils.SetupStreamingPassthrough(ctx, resp) { responseChan := make(chan *schemas.BifrostStreamChunk) @@ -932,8 +934,6 @@ func (provider *ReplicateProvider) ChatCompletionStream(ctx *schemas.BifrostCont // Create response channel responseChan := make(chan *schemas.BifrostStreamChunk, schemas.DefaultStreamBufferSize) - providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) - // Start streaming in a goroutine go func() { defer providerUtils.EnsureStreamFinalizerCalled(ctx, postHookSpanFinalizer) @@ -985,7 +985,7 @@ func (provider *ReplicateProvider) ChatCompletionStream(ctx *schemas.BifrostCont if readErr != io.EOF { ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) provider.logger.Warn("Error reading stream: %v", readErr) - enrichedErr := providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, readErr), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + enrichedErr := providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, readErr), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, enrichedErr, responseChan, provider.logger, postHookSpanFinalizer) } break @@ -1060,7 +1060,7 @@ func (provider *ReplicateProvider) ChatCompletionStream(ctx *schemas.BifrostCont "prediction was canceled", fmt.Errorf("stream ended: prediction canceled")) ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - enrichedErr := providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + enrichedErr := providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, enrichedErr, responseChan, provider.logger, postHookSpanFinalizer) // Explicitly close the body stream to terminate connection to Replicate resp.CloseBodyStream() @@ -1075,7 +1075,7 @@ func (provider *ReplicateProvider) ChatCompletionStream(ctx *schemas.BifrostCont errorMsg, fmt.Errorf("stream ended with error")) ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - enrichedErr := providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + enrichedErr := providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, enrichedErr, responseChan, provider.logger, postHookSpanFinalizer) // Explicitly close the body stream to terminate connection to Replicate resp.CloseBodyStream() @@ -1164,7 +1164,7 @@ func (provider *ReplicateProvider) Responses(ctx *schemas.BifrostContext, key sc provider.sendBackRawResponse, ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // if not sync, poll until done @@ -1179,13 +1179,13 @@ func (provider *ReplicateProvider) Responses(ctx *schemas.BifrostContext, key sc providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } } // Check for terminal error status (failed/canceled) after sync mode or polling if err := checkForErrorStatus(prediction); err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if providerResponseHeaders != nil { @@ -1239,7 +1239,7 @@ func (provider *ReplicateProvider) ResponsesStream(ctx *schemas.BifrostContext, startTime := time.Now() // Create prediction - prediction, _, _, _, err := createPrediction( + prediction, _, latency, _, err := createPrediction( ctx, provider.client, jsonData, @@ -1252,7 +1252,7 @@ func (provider *ReplicateProvider) ResponsesStream(ctx *schemas.BifrostContext, provider.sendBackRawResponse, ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Verify stream URL is available @@ -1260,7 +1260,7 @@ func (provider *ReplicateProvider) ResponsesStream(ctx *schemas.BifrostContext, bifrostErr := providerUtils.NewBifrostOperationError( "stream URL not available in prediction response", fmt.Errorf("prediction response missing stream URL")) - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } streamURL := *prediction.URLs.Stream @@ -1284,7 +1284,9 @@ func (provider *ReplicateProvider) ResponsesStream(ctx *schemas.BifrostContext, providerUtils.SetExtraHeaders(ctx, req, provider.networkConfig.ExtraHeaders, nil) // Make the streaming request + startTime = time.Now() streamErr := provider.streamingClient.Do(req, resp) + latency = time.Since(startTime) if streamErr != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(streamErr, context.Canceled) { @@ -1295,12 +1297,12 @@ func (provider *ReplicateProvider) ResponsesStream(ctx *schemas.BifrostContext, Message: schemas.ErrRequestCancelled, Error: streamErr, }, - }, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + }, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if errors.Is(streamErr, fasthttp.ErrTimeout) || errors.Is(streamErr, context.DeadlineExceeded) { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, streamErr), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, streamErr), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, streamErr), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, streamErr), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Extract provider response headers before status check so error responses also forward them @@ -1310,9 +1312,11 @@ func (provider *ReplicateProvider) ResponsesStream(ctx *schemas.BifrostContext, if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) body := resp.Body() - return nil, providerUtils.EnrichError(ctx, parseReplicateError(body, resp.StatusCode()), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseReplicateError(body, resp.StatusCode()), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } + providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) + // Large payload streaming passthrough — pipe raw upstream SSE to client if providerUtils.SetupStreamingPassthrough(ctx, resp) { responseChan := make(chan *schemas.BifrostStreamChunk) @@ -1323,8 +1327,6 @@ func (provider *ReplicateProvider) ResponsesStream(ctx *schemas.BifrostContext, // Create response channel responseChan := make(chan *schemas.BifrostStreamChunk, schemas.DefaultStreamBufferSize) - providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) - // Start streaming in a goroutine go func() { // Registered first so the post-hook span finalizer runs on every exit @@ -1350,7 +1352,7 @@ func (provider *ReplicateProvider) ResponsesStream(ctx *schemas.BifrostContext, "provider returned an empty response", fmt.Errorf("provider returned an empty response")) ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse), responseChan, provider.logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency), responseChan, provider.logger, postHookSpanFinalizer) return } @@ -1401,7 +1403,7 @@ func (provider *ReplicateProvider) ResponsesStream(ctx *schemas.BifrostContext, bifrostErr.ExtraFields.RawResponse = rawResponseChunks } - enrichedErr := providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + enrichedErr := providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, enrichedErr, responseChan, provider.logger, postHookSpanFinalizer) return } @@ -1682,7 +1684,7 @@ func (provider *ReplicateProvider) ResponsesStream(ctx *schemas.BifrostContext, } ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - enrichedErr := providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + enrichedErr := providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, enrichedErr, responseChan, provider.logger, postHookSpanFinalizer) resp.CloseBodyStream() return @@ -1773,7 +1775,7 @@ func (provider *ReplicateProvider) ImageGeneration(ctx *schemas.BifrostContext, provider.sendBackRawResponse, ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // If async mode and not complete, poll until done @@ -1788,13 +1790,13 @@ func (provider *ReplicateProvider) ImageGeneration(ctx *schemas.BifrostContext, providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } } // Check for terminal error status (failed/canceled) after sync mode or polling if err := checkForErrorStatus(prediction); err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if providerResponseHeaders != nil { @@ -1804,7 +1806,7 @@ func (provider *ReplicateProvider) ImageGeneration(ctx *schemas.BifrostContext, // Convert to Bifrost response bifrostResponse, err := ToBifrostImageGenerationResponse(prediction) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Set extra fields @@ -1854,7 +1856,7 @@ func (provider *ReplicateProvider) ImageGenerationStream(ctx *schemas.BifrostCon ) startTime := time.Now() // Create prediction - prediction, _, _, _, err := createPrediction( + prediction, _, latency, _, err := createPrediction( ctx, provider.client, jsonData, @@ -1867,7 +1869,7 @@ func (provider *ReplicateProvider) ImageGenerationStream(ctx *schemas.BifrostCon provider.sendBackRawResponse, ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Verify stream URL is available @@ -1882,6 +1884,7 @@ func (provider *ReplicateProvider) ImageGenerationStream(ctx *schemas.BifrostCon nil, sendBackRawRequest, sendBackRawResponse, + latency, ) } @@ -1896,6 +1899,8 @@ func (provider *ReplicateProvider) ImageGenerationStream(ctx *schemas.BifrostCon // Store provider response headers in context for transport layer ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerUtils.ExtractProviderResponseHeaders(resp)) + providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) + // Large payload streaming passthrough — pipe raw upstream SSE to client if providerUtils.SetupStreamingPassthrough(ctx, resp) { responseChan := make(chan *schemas.BifrostStreamChunk) @@ -1906,8 +1911,6 @@ func (provider *ReplicateProvider) ImageGenerationStream(ctx *schemas.BifrostCon // Create response channel responseChan := make(chan *schemas.BifrostStreamChunk, schemas.DefaultStreamBufferSize) - providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) - // Start streaming in a goroutine go func() { defer providerUtils.EnsureStreamFinalizerCalled(ctx, postHookSpanFinalizer) @@ -1961,7 +1964,7 @@ func (provider *ReplicateProvider) ImageGenerationStream(ctx *schemas.BifrostCon if readErr != io.EOF { ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) provider.logger.Warn(fmt.Sprintf("Error reading SSE stream: %v", readErr)) - enrichedErr := providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, readErr), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + enrichedErr := providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, readErr), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, enrichedErr, responseChan, provider.logger, postHookSpanFinalizer) } break @@ -2047,7 +2050,7 @@ func (provider *ReplicateProvider) ImageGenerationStream(ctx *schemas.BifrostCon if sendBackRawResponse && len(rawResponseChunks) > 0 { bifrostErr.ExtraFields.RawResponse = rawResponseChunks } - bifrostErr = providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + bifrostErr = providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, bifrostErr, responseChan, provider.logger, postHookSpanFinalizer) return @@ -2059,7 +2062,7 @@ func (provider *ReplicateProvider) ImageGenerationStream(ctx *schemas.BifrostCon if sendBackRawResponse && len(rawResponseChunks) > 0 { bifrostErr.ExtraFields.RawResponse = rawResponseChunks } - bifrostErr = providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + bifrostErr = providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, bifrostErr, responseChan, provider.logger, postHookSpanFinalizer) return @@ -2123,7 +2126,7 @@ func (provider *ReplicateProvider) ImageGenerationStream(ctx *schemas.BifrostCon rawResponseChunks = append(rawResponseChunks, ReplicateSSEEvent{Event: eventType, Data: eventData}) bifrostErr.ExtraFields.RawResponse = rawResponseChunks } - bifrostErr = providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + bifrostErr = providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, bifrostErr, responseChan, provider.logger, postHookSpanFinalizer) return @@ -2178,7 +2181,7 @@ func (provider *ReplicateProvider) ImageEdit(ctx *schemas.BifrostContext, key sc provider.sendBackRawResponse, ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // If async mode and not complete, poll until done @@ -2193,13 +2196,13 @@ func (provider *ReplicateProvider) ImageEdit(ctx *schemas.BifrostContext, key sc providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } } // Check for terminal error status (failed/canceled) after sync mode or polling if err := checkForErrorStatus(prediction); err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if providerResponseHeaders != nil { @@ -2209,7 +2212,7 @@ func (provider *ReplicateProvider) ImageEdit(ctx *schemas.BifrostContext, key sc // Convert to Bifrost response (reuse image generation response format) bifrostResponse, err := ToBifrostImageGenerationResponse(prediction) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Set extra fields @@ -2260,7 +2263,7 @@ func (provider *ReplicateProvider) ImageEditStream(ctx *schemas.BifrostContext, startTime := time.Now() // Create prediction - prediction, _, _, _, err := createPrediction( + prediction, _, latency, _, err := createPrediction( ctx, provider.client, jsonData, @@ -2273,7 +2276,7 @@ func (provider *ReplicateProvider) ImageEditStream(ctx *schemas.BifrostContext, provider.sendBackRawResponse, ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Verify stream URL is available @@ -2288,6 +2291,7 @@ func (provider *ReplicateProvider) ImageEditStream(ctx *schemas.BifrostContext, nil, sendBackRawRequest, sendBackRawResponse, + latency, ) } @@ -2302,6 +2306,8 @@ func (provider *ReplicateProvider) ImageEditStream(ctx *schemas.BifrostContext, // Store provider response headers in context for transport layer ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerUtils.ExtractProviderResponseHeaders(resp)) + providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) + // Large payload streaming passthrough — pipe raw upstream SSE to client if providerUtils.SetupStreamingPassthrough(ctx, resp) { responseChan := make(chan *schemas.BifrostStreamChunk) @@ -2312,8 +2318,6 @@ func (provider *ReplicateProvider) ImageEditStream(ctx *schemas.BifrostContext, // Create response channel responseChan := make(chan *schemas.BifrostStreamChunk, schemas.DefaultStreamBufferSize) - providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) - // Start streaming in a goroutine go func() { defer providerUtils.EnsureStreamFinalizerCalled(ctx, postHookSpanFinalizer) @@ -2365,7 +2369,7 @@ func (provider *ReplicateProvider) ImageEditStream(ctx *schemas.BifrostContext, return } if readErr != io.EOF { - enrichedErr := providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("stream read error", readErr), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + enrichedErr := providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("stream read error", readErr), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, enrichedErr, responseChan, provider.logger, postHookSpanFinalizer) } @@ -2449,7 +2453,7 @@ func (provider *ReplicateProvider) ImageEditStream(ctx *schemas.BifrostContext, if sendBackRawResponse && len(rawResponseChunks) > 0 { bifrostErr.ExtraFields.RawResponse = rawResponseChunks } - bifrostErr = providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + bifrostErr = providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, bifrostErr, responseChan, provider.logger, postHookSpanFinalizer) return @@ -2460,7 +2464,7 @@ func (provider *ReplicateProvider) ImageEditStream(ctx *schemas.BifrostContext, if sendBackRawResponse && len(rawResponseChunks) > 0 { bifrostErr.ExtraFields.RawResponse = rawResponseChunks } - bifrostErr = providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + bifrostErr = providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, bifrostErr, responseChan, provider.logger, postHookSpanFinalizer) return @@ -2509,7 +2513,7 @@ func (provider *ReplicateProvider) ImageEditStream(ctx *schemas.BifrostContext, rawResponseChunks = append(rawResponseChunks, ReplicateSSEEvent{Event: eventType, Data: eventData}) bifrostErr.ExtraFields.RawResponse = rawResponseChunks } - bifrostErr = providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + bifrostErr = providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, bifrostErr, responseChan, provider.logger, postHookSpanFinalizer) return @@ -2566,7 +2570,7 @@ func (provider *ReplicateProvider) VideoGeneration(ctx *schemas.BifrostContext, provider.sendBackRawResponse, ) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if providerResponseHeaders != nil { @@ -2576,7 +2580,7 @@ func (provider *ReplicateProvider) VideoGeneration(ctx *schemas.BifrostContext, // Convert to Bifrost response bifrostResponse, err := ToBifrostVideoGenerationResponse(prediction) if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse.ID = providerUtils.AddVideoIDProviderSuffix(bifrostResponse.ID, schemas.Replicate) @@ -2635,6 +2639,7 @@ func (provider *ReplicateProvider) VideoRetrieve(ctx *schemas.BifrostContext, ke nil, provider.sendBackRawRequest, provider.sendBackRawResponse, + latency, ) } @@ -2655,7 +2660,7 @@ func (provider *ReplicateProvider) VideoRetrieve(ctx *schemas.BifrostContext, ke bifrostResponse, convertErr := ToBifrostVideoGenerationResponse(&prediction) if convertErr != nil { - return nil, providerUtils.EnrichError(ctx, convertErr, nil, body, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, convertErr, nil, body, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse.ID = providerUtils.AddVideoIDProviderSuffix(bifrostResponse.ID, providerName) @@ -2716,9 +2721,9 @@ func (provider *ReplicateProvider) VideoDownload(ctx *schemas.BifrostContext, ke return nil, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK { - return nil, providerUtils.NewBifrostOperationError( + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostOperationError( fmt.Sprintf("failed to download video: HTTP %d", resp.StatusCode()), - nil) + nil), latency) } providerResponseHeaders := providerUtils.ExtractProviderResponseHeaders(resp) @@ -2904,7 +2909,7 @@ func (provider *ReplicateProvider) FileUpload(ctx *schemas.BifrostContext, key s // Handle error response if resp.StatusCode() != fasthttp.StatusOK && resp.StatusCode() != fasthttp.StatusCreated { provider.logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) - return nil, parseReplicateError(resp.Body(), resp.StatusCode()) + return nil, providerUtils.SetErrorLatency(parseReplicateError(resp.Body(), resp.StatusCode()), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -2993,7 +2998,7 @@ func (provider *ReplicateProvider) FileList(ctx *schemas.BifrostContext, keys [] // Handle error response if resp.StatusCode() != fasthttp.StatusOK { provider.logger.Debug("error from %s provider: %s", providerName, string(resp.Body())) - return nil, parseReplicateError(resp.Body(), resp.StatusCode()) + return nil, providerUtils.SetErrorLatency(parseReplicateError(resp.Body(), resp.StatusCode()), latency) } body, decodeErr := providerUtils.CheckAndDecodeBody(resp) diff --git a/core/providers/replicate/utils.go b/core/providers/replicate/utils.go index 336dbbb0578..4bff42293e3 100644 --- a/core/providers/replicate/utils.go +++ b/core/providers/replicate/utils.go @@ -9,6 +9,7 @@ import ( "regexp" "strconv" "strings" + "time" providerUtils "github.com/maximhq/bifrost/core/providers/utils" schemas "github.com/maximhq/bifrost/core/schemas" @@ -106,25 +107,27 @@ func listenToReplicateStreamURL( } // Make request + startTime := time.Now() err := client.Do(req, resp) + latency := time.Since(startTime) fasthttp.ReleaseRequest(req) if err != nil { providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { - return nil, nil, &schemas.BifrostError{ + return nil, nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } - return nil, nil, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, err) + return nil, nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, err), latency) } // Extract provider response headers before status check so error responses also forward them @@ -135,7 +138,7 @@ func listenToReplicateStreamURL( // Check for HTTP errors if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) - return nil, nil, parseReplicateError(resp.Body(), resp.StatusCode()) + return nil, nil, providerUtils.SetErrorLatency(parseReplicateError(resp.Body(), resp.StatusCode()), latency) } return resp.BodyStream(), resp, nil diff --git a/core/providers/runware/runware.go b/core/providers/runware/runware.go index 37aa41dce06..9ead5149ee9 100644 --- a/core/providers/runware/runware.go +++ b/core/providers/runware/runware.go @@ -191,14 +191,14 @@ func (provider *RunwareProvider) handleImageInference(ctx *schemas.BifrostContex // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, providerUtils.EnrichError(ctx, parseRunwareError(resp), body, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseRunwareError(resp), body, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Decode response body respBody, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { rawErrBody := append([]byte(nil), resp.Body()...) - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), body, rawErrBody, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), body, rawErrBody, sendBackRawRequest, sendBackRawResponse, latency) } // Parse response envelope @@ -211,7 +211,7 @@ func (provider *RunwareProvider) handleImageInference(ctx *schemas.BifrostContex // Convert to Bifrost response bifrostResp, bifrostErr := ToBifrostImageGenerationResponse(&runwareResp) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, body, respBody, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, body, respBody, sendBackRawRequest, sendBackRawResponse, latency) } bifrostResp.Model = model @@ -277,14 +277,14 @@ func (provider *RunwareProvider) sendTaskArray(ctx *schemas.BifrostContext, key lat, bErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bErr != nil { - return reqBody, nil, 0, bErr + return reqBody, nil, lat, bErr } if resp.StatusCode() != fasthttp.StatusOK { - return reqBody, nil, 0, parseRunwareError(resp) + return reqBody, nil, lat, providerUtils.SetErrorLatency(parseRunwareError(resp), lat) } decoded, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { - return reqBody, nil, 0, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err) + return reqBody, nil, lat, providerUtils.SetErrorLatency(providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), lat) } // Copy out: the fasthttp response buffer is released when this function returns. return reqBody, append([]byte(nil), decoded...), lat, nil @@ -309,7 +309,7 @@ func (provider *RunwareProvider) VideoGeneration(ctx *schemas.BifrostContext, ke reqBody, respBody, latency, bifrostErr := provider.sendTaskArray(ctx, key, jsonData) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, reqBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, reqBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } var videoResp RunwareResponse @@ -320,7 +320,7 @@ func (provider *RunwareProvider) VideoGeneration(ctx *schemas.BifrostContext, ke result, bifrostErr := firstVideoResult(&videoResp) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, reqBody, respBody, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, reqBody, respBody, sendBackRawRequest, sendBackRawResponse, latency) } bifrostResp := ToBifrostVideoGenerationResponse(result) @@ -351,7 +351,7 @@ func (provider *RunwareProvider) VideoRetrieve(ctx *schemas.BifrostContext, key reqBody, respBody, latency, bifrostErr := provider.sendTaskArray(ctx, key, jsonData) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, reqBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, reqBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } var videoResp RunwareResponse @@ -362,7 +362,7 @@ func (provider *RunwareProvider) VideoRetrieve(ctx *schemas.BifrostContext, key result, bifrostErr := firstVideoResult(&videoResp) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, reqBody, respBody, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, reqBody, respBody, sendBackRawRequest, sendBackRawResponse, latency) } bifrostResp := ToBifrostVideoGenerationResponse(result) @@ -405,7 +405,7 @@ func (provider *RunwareProvider) VideoDownload(ctx *schemas.BifrostContext, key return nil, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK { - return nil, providerUtils.NewBifrostOperationError(fmt.Sprintf("failed to download video: HTTP %d", resp.StatusCode()), nil) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostOperationError(fmt.Sprintf("failed to download video: HTTP %d", resp.StatusCode()), nil), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { diff --git a/core/providers/runway/runway.go b/core/providers/runway/runway.go index 77b0575c15f..e7313bf55c6 100644 --- a/core/providers/runway/runway.go +++ b/core/providers/runway/runway.go @@ -186,27 +186,27 @@ func (provider *RunwayProvider) HandleRunwayImageTask(ctx *schemas.BifrostContex // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, providerUtils.EnrichError(ctx, parseRunwayError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseRunwayError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Decode response body body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { rawErrBody := append([]byte(nil), resp.Body()...) - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, rawErrBody, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, rawErrBody, sendBackRawRequest, sendBackRawResponse, latency) } // Parse task creation response var taskResp RunwayTaskCreationResponse rawRequest, _, bifrostErr := providerUtils.HandleProviderResponse(body, &taskResp, jsonData, sendBackRawRequest, sendBackRawResponse) if bifrostErr != nil { - return nil, bifrostErr + return nil, providerUtils.SetErrorLatency(bifrostErr, latency) } // Poll the task until it reaches a terminal state taskDetails, rawResponse, bifrostErr := provider.pollRunwayTask(ctx, key, taskResp.ID, sendBackRawResponse) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Convert to Bifrost response @@ -276,25 +276,25 @@ func (provider *RunwayProvider) retrieveRunwayTask(ctx *schemas.BifrostContext, req.Header.Set("Authorization", "Bearer "+key.Value.GetValue()) } - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { return nil, nil, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK { - return nil, nil, parseRunwayError(resp) + return nil, nil, providerUtils.SetErrorLatency(parseRunwayError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { - return nil, nil, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err) + return nil, nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), latency) } var taskDetails RunwayTaskDetailsResponse _, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(body, &taskDetails, nil, false, sendBackRawResponse) if bifrostErr != nil { - return nil, nil, bifrostErr + return nil, nil, providerUtils.SetErrorLatency(bifrostErr, latency) } return &taskDetails, rawResponse, nil @@ -382,21 +382,21 @@ func (provider *RunwayProvider) VideoGeneration(ctx *schemas.BifrostContext, key // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, providerUtils.EnrichError(ctx, parseRunwayError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseRunwayError(resp), jsonData, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Decode response body body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { rawErrBody := append([]byte(nil), resp.Body()...) - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, rawErrBody, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), jsonData, rawErrBody, sendBackRawRequest, sendBackRawResponse, latency) } // Parse response var taskResp RunwayTaskCreationResponse rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(body, &taskResp, jsonData, sendBackRawRequest, sendBackRawResponse) if bifrostErr != nil { - return nil, bifrostErr + return nil, providerUtils.SetErrorLatency(bifrostErr, latency) } // Convert to Bifrost response @@ -453,21 +453,21 @@ func (provider *RunwayProvider) VideoRetrieve(ctx *schemas.BifrostContext, key s // Handle error response if resp.StatusCode() != fasthttp.StatusOK { - return nil, providerUtils.EnrichError(ctx, parseRunwayError(resp), nil, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseRunwayError(resp), nil, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Decode response body body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { rawErrBody := append([]byte(nil), resp.Body()...) - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), nil, rawErrBody, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), nil, rawErrBody, sendBackRawRequest, sendBackRawResponse, latency) } // Parse response var taskDetails RunwayTaskDetailsResponse rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(body, &taskDetails, nil, sendBackRawRequest, sendBackRawResponse) if bifrostErr != nil { - return nil, bifrostErr + return nil, providerUtils.SetErrorLatency(bifrostErr, latency) } // Convert to Bifrost response @@ -532,15 +532,15 @@ func (provider *RunwayProvider) VideoDownload(ctx *schemas.BifrostContext, key s return nil, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK { - return nil, providerUtils.NewBifrostOperationError( + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostOperationError( fmt.Sprintf("failed to download video: HTTP %d", resp.StatusCode()), - nil) + nil), latency) } // Get content and content type body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { rawErrBody := append([]byte(nil), resp.Body()...) - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), nil, rawErrBody, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseDecode, err), nil, rawErrBody, sendBackRawRequest, sendBackRawResponse, latency) } contentType := string(resp.Header.ContentType()) if contentType == "" { @@ -598,7 +598,7 @@ func (provider *RunwayProvider) VideoDelete(ctx *schemas.BifrostContext, key sch // Handle error response - Runway returns 204 No Content on success if resp.StatusCode() != fasthttp.StatusNoContent { - return nil, providerUtils.EnrichError(ctx, parseRunwayError(resp), nil, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseRunwayError(resp), nil, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Build response - Runway returns empty body on 204 diff --git a/core/providers/sgl/sgl.go b/core/providers/sgl/sgl.go index 3f31830d864..d36d664b36e 100644 --- a/core/providers/sgl/sgl.go +++ b/core/providers/sgl/sgl.go @@ -136,7 +136,7 @@ func (provider *SGLProvider) TextCompletion(ctx *schemas.BifrostContext, key sch provider.client, baseURL+providerUtils.GetPathFromContext(ctx, "/v1/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -156,16 +156,12 @@ func (provider *SGLProvider) TextCompletionStream(ctx *schemas.BifrostContext, p if bifrostErr != nil { return nil, bifrostErr } - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } return openai.HandleOpenAITextCompletionStreaming( ctx, provider.streamingClient, baseURL+providerUtils.GetPathFromContext(ctx, "/v1/completions"), request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -192,13 +188,14 @@ func (provider *SGLProvider) ChatCompletion(ctx *schemas.BifrostContext, key sch provider.client, baseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), nil, nil, + nil, provider.logger, ) } @@ -213,17 +210,12 @@ func (provider *SGLProvider) ChatCompletionStream(ctx *schemas.BifrostContext, p if bifrostErr != nil { return nil, bifrostErr } - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } - // Use shared OpenAI-compatible streaming logic return openai.HandleOpenAIChatCompletionStreaming( ctx, provider.streamingClient, baseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -235,6 +227,7 @@ func (provider *SGLProvider) ChatCompletionStream(ctx *schemas.BifrostContext, p nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -275,7 +268,7 @@ func (provider *SGLProvider) Embedding(ctx *schemas.BifrostContext, key schemas. provider.client, baseURL+providerUtils.GetPathFromContext(ctx, "/v1/embeddings"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), diff --git a/core/providers/utils/bodysigner.go b/core/providers/utils/bodysigner.go new file mode 100644 index 00000000000..69edad70c69 --- /dev/null +++ b/core/providers/utils/bodysigner.go @@ -0,0 +1,9 @@ +package utils + +import schemas "github.com/maximhq/bifrost/core/schemas" + +// BodySigner signs the final request body bytes after the handler has marshaled them and +// returns auth headers to set on the outgoing request. Handlers invoke it immediately after +// building the body, only when non-nil. The caller decides whether signing applies (e.g. AWS +// SigV4 for Bedrock Mantle when no API key is present); the handler stays auth-scheme-agnostic. +type BodySigner func(jsonData []byte) (map[string]string, *schemas.BifrostError) diff --git a/core/providers/utils/modelparamscache.go b/core/providers/utils/modelparamscache.go index 7ece70a7a00..8d45b139365 100644 --- a/core/providers/utils/modelparamscache.go +++ b/core/providers/utils/modelparamscache.go @@ -13,7 +13,7 @@ const DefaultModelParamsCacheSize = 2048 // ModelParams holds cached parameters for a model. // Add new fields here as more model-level parameters need caching. type ModelParams struct { - MaxOutputTokens *int + MaxOutputTokens *int IsVertexMultiRegionOnly *bool // true when model is only available on Vertex multi-region pool endpoints (rep.googleapis.com) } @@ -48,6 +48,7 @@ var ( // knownAnthropicMaxOutputTokens provides static fallback defaults for Claude models // when both cache and DB miss handler return nothing. Only Anthropic requires max_tokens. var knownAnthropicMaxOutputTokens = map[string]int{ + "claude-fable-5": 128000, "claude-opus-4-6": 128000, "claude-sonnet-5": 128000, "claude-sonnet-4-6": 64000, diff --git a/core/providers/utils/utils.go b/core/providers/utils/utils.go index 03822dae129..6ee9898856d 100644 --- a/core/providers/utils/utils.go +++ b/core/providers/utils/utils.go @@ -35,6 +35,22 @@ import ( "github.com/valyala/fasthttp/fasthttpproxy" ) +// ThoughtSignatureSeparator delimits a tool call's base ID from a provider reasoning +// signature embedded in the call_id (e.g. Gemini thoughtSignatures), formatted as +// "_ts_". +const ThoughtSignatureSeparator = "_ts_" + +// StripThoughtSignature returns the base tool-call ID without any embedded provider +// reasoning signature. It is deterministic, so a tool call and its matching output strip +// to the same ID. Providers that cannot use the signature (e.g. OpenAI, which caps call_id +// at 64 chars) call this before sending the ID upstream. +func StripThoughtSignature(callID string) string { + if base, _, found := strings.Cut(callID, ThoughtSignatureSeparator); found { + return base + } + return callID +} + // sortedAPI is a sonic encoder/decoder that sorts map keys during marshaling. // This ensures deterministic JSON output for map[string]interface{} values, // which is critical for LLM prompt caching (e.g., Anthropic cache keying). @@ -110,6 +126,15 @@ var UnsupportedSpeechStreamModels = []string{"tts-1", "tts-1-hd"} // noop is a reusable no-op function returned by MakeRequestWithContext on the normal path. var noop = func() {} +// SetErrorLatency stamps provider/request latency onto an error so downstream +// logging and client-facing error details can show timing even without a response. +func SetErrorLatency(bifrostErr *schemas.BifrostError, latency time.Duration) *schemas.BifrostError { + if bifrostErr != nil { + bifrostErr.ExtraFields.Latency = latency.Milliseconds() + } + return bifrostErr +} + // makeRequestWithDoFunc is the shared core behind MakeRequestWithContext and // MakeRequestWithContextFollowRedirects. It runs do() in a goroutine and handles // context cancellation, latency tracking, and error classification uniformly. @@ -152,6 +177,7 @@ func makeRequestWithDoFunc(ctx context.Context, do func() error) (time.Duration, Message: fmt.Sprintf("Request timed out by context: %v", ctx.Err()), Error: ctx.Err(), }, + ExtraFields: schemas.BifrostErrorExtraFields{Latency: latency.Milliseconds()}, }, func() { <-errChan } } statusCode := 499 @@ -164,6 +190,7 @@ func makeRequestWithDoFunc(ctx context.Context, do func() error) (time.Duration, Message: fmt.Sprintf("Request cancelled by context: %v", ctx.Err()), Error: ctx.Err(), }, + ExtraFields: schemas.BifrostErrorExtraFields{Latency: latency.Milliseconds()}, }, func() { <-errChan } case err := <-errChan: // The do() call completed. @@ -178,25 +205,26 @@ func makeRequestWithDoFunc(ctx context.Context, do func() error) (time.Duration, Message: schemas.ErrRequestCancelled, Error: err, }, + ExtraFields: schemas.BifrostErrorExtraFields{Latency: latency.Milliseconds()}, }, noop } // Check for timeout errors first before checking net.OpError to avoid misclassification. if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return latency, NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), noop + return latency, SetErrorLatency(NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency), noop } // Check if error implements net.Error and has Timeout() == true. var netErr net.Error if errors.As(err, &netErr) && netErr.Timeout() { - return latency, NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), noop + return latency, SetErrorLatency(NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency), noop } // Check for DNS lookup and network errors after timeout checks. var opErr *net.OpError var dnsErr *net.DNSError if errors.As(err, &opErr) || errors.As(err, &dnsErr) { - return latency, NewBifrostUpstreamConnectionError(schemas.ErrProviderNetworkError, err), noop + return latency, SetErrorLatency(NewBifrostUpstreamConnectionError(schemas.ErrProviderNetworkError, err), latency), noop } // The HTTP request itself failed (e.g., connection error, fasthttp timeout). - return latency, NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), noop + return latency, SetErrorLatency(NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), latency), noop } // HTTP request was successful from fasthttp's perspective (err is nil). // The caller should check resp.StatusCode() for HTTP-level errors (4xx, 5xx). @@ -210,13 +238,15 @@ func makeRequestWithDoFunc(ctx context.Context, do func() error) (time.Duration, // path it blocks until the background client.Do goroutine finishes, preventing a data race // between the still-running goroutine and the caller's release of req/resp. func MakeRequestWithContext(ctx context.Context, client *fasthttp.Client, req *fasthttp.Request, resp *fasthttp.Response) (time.Duration, *schemas.BifrostError, func()) { - return makeRequestWithDoFunc(ctx, func() error { return client.Do(req, resp) }) + latency, bifrostErr, wait := makeRequestWithDoFunc(ctx, func() error { return client.Do(req, resp) }) + return latency, bifrostErr, wait } // MakeRequestWithContextFollowRedirects is like MakeRequestWithContext but follows up to // maxRedirects HTTP redirects automatically (equivalent to curl's -L flag). func MakeRequestWithContextFollowRedirects(ctx context.Context, client *fasthttp.Client, req *fasthttp.Request, resp *fasthttp.Response, maxRedirects int) (time.Duration, *schemas.BifrostError, func()) { - return makeRequestWithDoFunc(ctx, func() error { return client.DoRedirects(req, resp, maxRedirects) }) + latency, bifrostErr, wait := makeRequestWithDoFunc(ctx, func() error { return client.DoRedirects(req, resp, maxRedirects) }) + return latency, bifrostErr, wait } // Deprecated: ConfigureRetry is now handled internally by ConfigureDialer. @@ -1428,11 +1458,16 @@ func EnrichError( responseBody []byte, sendBackRawRequest bool, sendBackRawResponse bool, + latency ...time.Duration, ) *schemas.BifrostError { if bifrostErr == nil { return bifrostErr } + if len(latency) > 0 { + SetErrorLatency(bifrostErr, latency[0]) + } + if ShouldSendBackRawRequest(ctx, sendBackRawRequest) && len(requestBody) > 0 { // Store as json.RawMessage to preserve exact JSON bytes (including key ordering). // Compact to remove insignificant whitespace that would break SSE framing. @@ -2129,9 +2164,8 @@ func ProcessAndSendResponse( streamResponse := BuildClientStreamChunk(ctx, processedResponse, processedError) - if !GateSendChunk(ctx, streamResponse, responseChan) { - return - } + // Complete the final-chunk span even if the client send fails, so a dropped connection can't strand it. + GateSendChunk(ctx, streamResponse, responseChan) // Check if this is the final chunk and complete deferred span with post-processed data if isFinalChunk := ctx.Value(schemas.BifrostContextKeyStreamEndIndicator); isFinalChunk != nil { @@ -2192,6 +2226,13 @@ func ProcessAndSendBifrostError( // path — e.g. a panic mid-stream — which would otherwise leak the plugin // pipeline back-reference held by the finalizer closure. // +// It also completes any deferred LLM span still parked for this trace so a +// goroutine that died before completing it cannot leak the span — and the +// accumulated response it pins. The stream-end indicator sets the status: an +// unended stream is marked failed, an ended one keeps its success status. +// completeDeferredSpan is idempotent (nil-handle guard), so this is a noop when +// the terminal path already cleared it. +// // Panics inside the finalizer are recovered and logged so they never mask an // in-flight panic that triggered the defer. func EnsureStreamFinalizerCalled(ctx context.Context, finalizer func(context.Context)) { @@ -2200,6 +2241,19 @@ func EnsureStreamFinalizerCalled(ctx context.Context, finalizer func(context.Con getLogger().Debug("recovered panic in deferred stream finalizer: %v", r) } }() + + // Complete any span the terminal path left parked. Unended = died mid-flight + // (mark failed); ended = delivery failed after success (keep the OK status). + if bfCtx, ok := ctx.(*schemas.BifrostContext); ok { + var streamErr *schemas.BifrostError + if ended, _ := bfCtx.Value(schemas.BifrostContextKeyStreamEndIndicator).(bool); !ended { + streamErr = &schemas.BifrostError{ + Error: &schemas.ErrorField{Message: "stream ended before completion"}, + } + } + completeDeferredSpan(bfCtx, nil, streamErr, finalizer) + } + if finalizer == nil { return } diff --git a/core/providers/utils/utils_test.go b/core/providers/utils/utils_test.go index 517da6bbaf3..12c68eca835 100644 --- a/core/providers/utils/utils_test.go +++ b/core/providers/utils/utils_test.go @@ -229,6 +229,44 @@ func TestEnrichError_OverwritesWithProvidedResponse(t *testing.T) { t.Log("✓ EnrichError sets RawRequest and RawResponse from provided bodies") } +func TestEnrichError_SetsLatency(t *testing.T) { + ctx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline) + bifrostErr := &schemas.BifrostError{ + IsBifrostError: false, + Error: &schemas.ErrorField{ + Message: "provider failed", + }, + } + + enrichedErr := EnrichError(ctx, bifrostErr, nil, nil, false, false, 42*time.Millisecond) + + if enrichedErr == nil { + t.Fatal("EnrichError() returned nil") + } + if enrichedErr.ExtraFields.Latency != 42 { + t.Fatalf("latency = %d, want 42", enrichedErr.ExtraFields.Latency) + } +} + +func TestEnrichError_DoesNotSetLatencyWithoutExplicitValue(t *testing.T) { + ctx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline) + bifrostErr := &schemas.BifrostError{ + IsBifrostError: false, + Error: &schemas.ErrorField{ + Message: "provider failed", + }, + } + + enrichedErr := EnrichError(ctx, bifrostErr, nil, nil, false, false) + + if enrichedErr == nil { + t.Fatal("EnrichError() returned nil") + } + if enrichedErr.ExtraFields.Latency != 0 { + t.Fatalf("latency = %d, want 0", bifrostErr.ExtraFields.Latency) + } +} + // TestEnrichError_RespectsFlags verifies that EnrichError respects // sendBackRawRequest and sendBackRawResponse flags func TestEnrichError_RespectsFlags(t *testing.T) { @@ -1878,3 +1916,166 @@ func TestExtractPassthroughProviderResponseHeaders(t *testing.T) { t.Fatalf("benign header x-request-id was dropped: %v", headers) } } + +func TestStripThoughtSignature(t *testing.T) { + cases := []struct { + name string + in string + want string + }{ + {"no separator", "call_abc123", "call_abc123"}, + {"gemini embedded signature", "search_ts_QUJDREVG", "search"}, + {"base id is also a gemini id", "fc_123_ts_QUJD", "fc_123"}, + {"separator only", "_ts_QUJD", ""}, + {"empty", "", ""}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if got := StripThoughtSignature(tc.in); got != tc.want { + t.Errorf("StripThoughtSignature(%q) = %q, want %q", tc.in, got, tc.want) + } + }) + } +} + +// finalizerTestTracer is a minimal schemas.Tracer that models only the +// deferred-span lifecycle: a span stays parked until ClearDeferredSpan runs. +// It records the status passed to EndSpan so tests can assert span outcomes. +type finalizerTestTracer struct { + schemas.NoOpTracer + parked bool + endStatus schemas.SpanStatus +} + +func (t *finalizerTestTracer) GetDeferredSpanHandle(_ string) schemas.SpanHandle { + if t.parked { + return struct{}{} // any non-nil handle + } + return nil +} + +func (t *finalizerTestTracer) ClearDeferredSpan(_ string) { t.parked = false } + +func (t *finalizerTestTracer) EndSpan(_ schemas.SpanHandle, status schemas.SpanStatus, _ string) { + t.endStatus = status +} + +// A streaming goroutine that exits without reaching the final-chunk path (a +// failed final send with a live context, or a mid-stream death) must not leak +// its deferred span. EnsureStreamFinalizerCalled runs on every goroutine exit +// and clears it. +func TestEnsureStreamFinalizerCalled_ClearsOrphanedDeferredSpan(t *testing.T) { + tracer := &finalizerTestTracer{parked: true} + + ctx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline) + ctx.SetValue(schemas.BifrostContextKeyTracer, tracer) + ctx.SetValue(schemas.BifrostContextKeyTraceID, "trace-1") + + if tracer.GetDeferredSpanHandle("trace-1") == nil { + t.Fatal("expected a parked deferred span before the finalizer runs") + } + + EnsureStreamFinalizerCalled(ctx, func(context.Context) {}) + + if tracer.GetDeferredSpanHandle("trace-1") != nil { + t.Error("deferred span should be cleared when the streaming goroutine exits") + } +} + +// The clear must survive a nil finalizer (finalizer is optional; the span +// cleanup is not). +func TestEnsureStreamFinalizerCalled_ClearsDeferredSpanWithNilFinalizer(t *testing.T) { + tracer := &finalizerTestTracer{parked: true} + + ctx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline) + ctx.SetValue(schemas.BifrostContextKeyTracer, tracer) + ctx.SetValue(schemas.BifrostContextKeyTraceID, "trace-1") + + EnsureStreamFinalizerCalled(ctx, nil) + + if tracer.GetDeferredSpanHandle("trace-1") != nil { + t.Error("deferred span should be cleared even when no finalizer is registered") + } +} + +// When the terminal path already cleared the span (the common case), the +// safety-net completion is a no-op (handle == nil early return) but the +// finalizer must still run exactly once. +func TestEnsureStreamFinalizerCalled_NoParkedSpan_StillRunsFinalizerOnce(t *testing.T) { + tracer := &finalizerTestTracer{parked: false} + + ctx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline) + ctx.SetValue(schemas.BifrostContextKeyTracer, tracer) + ctx.SetValue(schemas.BifrostContextKeyTraceID, "trace-1") + + calls := 0 + EnsureStreamFinalizerCalled(ctx, func(context.Context) { calls++ }) + + if calls != 1 { + t.Errorf("finalizer should run exactly once on the no-op path, ran %d times", calls) + } +} + +// A stream that exits with its span parked and no terminal chunk +// (StreamEndIndicator unset) died mid-flight and must be marked failed — not OK. +func TestEnsureStreamFinalizerCalled_IncompleteStreamMarkedError(t *testing.T) { + tracer := &finalizerTestTracer{parked: true} + + ctx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline) + ctx.SetValue(schemas.BifrostContextKeyTracer, tracer) + ctx.SetValue(schemas.BifrostContextKeyTraceID, "trace-1") + // StreamEndIndicator deliberately unset: the stream never reached its end. + + EnsureStreamFinalizerCalled(ctx, func(context.Context) {}) + + if tracer.endStatus != schemas.SpanStatusError { + t.Errorf("incomplete stream should end as %q, got %q", schemas.SpanStatusError, tracer.endStatus) + } +} + +// A stream that reached its terminal chunk (StreamEndIndicator set) but was left +// parked — e.g. the final send failed — succeeded at the LLM level and must keep +// its OK status, never a fabricated error. +func TestEnsureStreamFinalizerCalled_CompletedStreamKeepsOkStatus(t *testing.T) { + tracer := &finalizerTestTracer{parked: true} + + ctx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline) + ctx.SetValue(schemas.BifrostContextKeyTracer, tracer) + ctx.SetValue(schemas.BifrostContextKeyTraceID, "trace-1") + ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) + + EnsureStreamFinalizerCalled(ctx, func(context.Context) {}) + + if tracer.endStatus != schemas.SpanStatusOk { + t.Errorf("completed-but-undelivered stream should end as %q, got %q", schemas.SpanStatusOk, tracer.endStatus) + } +} + +// Fix A: when the final chunk's send fails with the context still alive (a closed +// consumer channel), ProcessAndSendResponse must still complete the deferred span +// with its real (success) outcome rather than strand it. +func TestProcessAndSendResponse_CompletesSpanWhenFinalSendFails(t *testing.T) { + tracer := &finalizerTestTracer{parked: true} + + ctx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline) + ctx.SetValue(schemas.BifrostContextKeyTracer, tracer) + ctx.SetValue(schemas.BifrostContextKeyTraceID, "trace-1") + ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) // final chunk + + // A closed channel makes GateSendChunk fail while the context is still alive. + responseChan := make(chan *schemas.BifrostStreamChunk) + close(responseChan) + + passthrough := func(_ *schemas.BifrostContext, resp *schemas.BifrostResponse, err *schemas.BifrostError) (*schemas.BifrostResponse, *schemas.BifrostError) { + return resp, err + } + + ProcessAndSendResponse(ctx, passthrough, &schemas.BifrostResponse{}, responseChan, func(context.Context) {}) + + if tracer.GetDeferredSpanHandle("trace-1") != nil { + t.Error("deferred span must be completed even when the final chunk send fails") + } + if tracer.endStatus != schemas.SpanStatusOk { + t.Errorf("successful stream whose delivery failed should end as %q, got %q", schemas.SpanStatusOk, tracer.endStatus) + } +} diff --git a/core/providers/vertex/cachedcontents.go b/core/providers/vertex/cachedcontents.go index 05adb24bcbe..4cd87352005 100644 --- a/core/providers/vertex/cachedcontents.go +++ b/core/providers/vertex/cachedcontents.go @@ -178,7 +178,7 @@ func (provider *VertexProvider) CachedContentCreate(ctx *schemas.BifrostContext, return nil, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK { - return nil, parseVertexCachedContentError(resp) + return nil, providerUtils.SetErrorLatency(parseVertexCachedContentError(resp), latency) } respBody, decErr := providerUtils.CheckAndDecodeBody(resp) @@ -250,7 +250,7 @@ func (provider *VertexProvider) cachedContentListByKey(ctx *schemas.BifrostConte return nil, latency, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK { - return nil, latency, parseVertexCachedContentError(resp) + return nil, latency, providerUtils.SetErrorLatency(parseVertexCachedContentError(resp), latency) } respBody, decErr := providerUtils.CheckAndDecodeBody(resp) @@ -323,7 +323,7 @@ func (provider *VertexProvider) cachedContentRetrieveByKey(ctx *schemas.BifrostC return nil, latency, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK { - return nil, latency, parseVertexCachedContentError(resp) + return nil, latency, providerUtils.SetErrorLatency(parseVertexCachedContentError(resp), latency) } respBody, decErr := providerUtils.CheckAndDecodeBody(resp) @@ -427,7 +427,7 @@ func (provider *VertexProvider) cachedContentUpdateByKey(ctx *schemas.BifrostCon return nil, latency, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK { - return nil, latency, parseVertexCachedContentError(resp) + return nil, latency, providerUtils.SetErrorLatency(parseVertexCachedContentError(resp), latency) } respBody, decErr := providerUtils.CheckAndDecodeBody(resp) @@ -512,7 +512,7 @@ func (provider *VertexProvider) cachedContentDeleteByKey(ctx *schemas.BifrostCon return nil, latency, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK { - return nil, latency, parseVertexCachedContentError(resp) + return nil, latency, providerUtils.SetErrorLatency(parseVertexCachedContentError(resp), latency) } return &schemas.BifrostCachedContentDeleteResponse{ diff --git a/core/providers/vertex/errors.go b/core/providers/vertex/errors.go index e0ed7f1d3d1..e91c3c5bc44 100644 --- a/core/providers/vertex/errors.go +++ b/core/providers/vertex/errors.go @@ -49,8 +49,14 @@ func parseVertexError(resp *fasthttp.Response) *schemas.BifrostError { return bifrostErr } - createError := func(message string) *schemas.BifrostError { + createError := func(message, status string) *schemas.BifrostError { bifrostErr := providerUtils.NewProviderAPIError(message, nil, resp.StatusCode(), nil, nil) + if status != "" { + if bifrostErr.Error == nil { + bifrostErr.Error = &schemas.ErrorField{} + } + bifrostErr.Error.Type = &status + } var rawResponse interface{} if err := sonic.Unmarshal(decodedBody, &rawResponse); err != nil { rawResponse = string(decodedBody) @@ -72,17 +78,27 @@ func parseVertexError(resp *fasthttp.Response) *schemas.BifrostError { return bifrostErr } if len(validationErr.Detail) > 0 { - return createError(validationErr.Detail[0].Msg) + return createError(validationErr.Detail[0].Msg, "") } - return createError("Unknown error") + return createError("Unknown error", "") } - return createError(vertexErr.Error.Message) + return createError(vertexErr.Error.Message, vertexErr.Error.Status) } if len(vertexErr) > 0 { - return createError(vertexErr[0].Error.Message) + return createError(vertexErr[0].Error.Message, vertexErr[0].Error.Status) + } + return createError("Unknown error", "") + } + // OpenAI error format succeeded with valid Error field. + openAIStatus := "" + if openAIErr.Error.Type != nil { + openAIStatus = *openAIErr.Error.Type + } + if openAIStatus == "" { + var single VertexError + if err := sonic.Unmarshal(decodedBody, &single); err == nil { + openAIStatus = single.Error.Status } - return createError("Unknown error") } - // OpenAI error format succeeded with valid Error field - return createError(openAIErr.Error.Message) + return createError(openAIErr.Error.Message, openAIStatus) } diff --git a/core/providers/vertex/errors_test.go b/core/providers/vertex/errors_test.go new file mode 100644 index 00000000000..d991994f64c --- /dev/null +++ b/core/providers/vertex/errors_test.go @@ -0,0 +1,40 @@ +package vertex + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "github.com/valyala/fasthttp" +) + +// TestParseVertexError_PopulatesStatusType verifies the Vertex status +// (e.g. RESOURCE_EXHAUSTED) is surfaced on error.type rather than being dropped, +// so passthrough/OpenAI-shaped consumers see the exception type. +func TestParseVertexError_PopulatesStatusType(t *testing.T) { + var resp fasthttp.Response + resp.SetStatusCode(fasthttp.StatusTooManyRequests) + resp.SetBodyString(`{"error":{"code":429,"message":"Quota exceeded","status":"RESOURCE_EXHAUSTED"}}`) + + bifrostErr := parseVertexError(&resp) + + require.NotNil(t, bifrostErr) + require.NotNil(t, bifrostErr.Error) + require.NotNil(t, bifrostErr.Error.Type, "nested error.type must be populated from status") + assert.Equal(t, "RESOURCE_EXHAUSTED", *bifrostErr.Error.Type) + assert.Equal(t, "Quota exceeded", bifrostErr.Error.Message) +} + +// TestParseVertexError_NoStatusNoType verifies that when the body carries no +// Vertex status we don't fabricate an error.type. +func TestParseVertexError_NoStatusNoType(t *testing.T) { + var resp fasthttp.Response + resp.SetStatusCode(fasthttp.StatusBadRequest) + resp.SetBodyString(`{"error":{"code":400,"message":"bad request"}}`) + + bifrostErr := parseVertexError(&resp) + + require.NotNil(t, bifrostErr) + require.NotNil(t, bifrostErr.Error) + assert.Nil(t, bifrostErr.Error.Type, "no status present, so none should be fabricated") +} diff --git a/core/providers/vertex/types.go b/core/providers/vertex/types.go index f155426bde3..1d9fc760771 100644 --- a/core/providers/vertex/types.go +++ b/core/providers/vertex/types.go @@ -9,8 +9,6 @@ import ( // Vertex AI Embedding API types const ( - DefaultVertexAnthropicVersion = "vertex-2023-10-16" - // VertexServiceTierHeader is the HTTP header used to request priority or flex processing on the global endpoint. VertexServiceTierHeader = "X-Vertex-AI-LLM-Shared-Request-Type" ) diff --git a/core/providers/vertex/utils.go b/core/providers/vertex/utils.go index 3db78980782..53b187390aa 100644 --- a/core/providers/vertex/utils.go +++ b/core/providers/vertex/utils.go @@ -4,7 +4,6 @@ import ( "fmt" "strings" - "github.com/maximhq/bifrost/core/providers/anthropic" "github.com/maximhq/bifrost/core/providers/gemini" providerUtils "github.com/maximhq/bifrost/core/providers/utils" schemas "github.com/maximhq/bifrost/core/schemas" @@ -59,38 +58,6 @@ func resolveVertexRegion(ctx *schemas.BifrostContext, key schemas.Key) string { return "" } -// getRequestBodyForAnthropicResponses serializes a BifrostResponsesRequest into the Anthropic wire format for Vertex AI. -// Compared to the native Anthropic path, it strips model/region fields, remaps tool versions, injects beta headers -// into the request body (rather than HTTP headers), and pins the Anthropic API version to DefaultVertexAnthropicVersion. -func getRequestBodyForAnthropicResponses(ctx *schemas.BifrostContext, request *schemas.BifrostResponsesRequest, deployment string, isStreaming bool, isCountTokens bool, betaHeaderOverrides map[string]bool, providerExtraHeaders map[string]string, shouldSendBackRawRequest bool, shouldSendBackRawResponse bool) ([]byte, *schemas.BifrostError) { - jsonBody, buildErr := anthropic.BuildAnthropicResponsesRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{ - Provider: schemas.Vertex, - Deployment: deployment, - DeleteModelField: true, - DeleteRegionField: true, - IsStreaming: isStreaming, - IsCountTokens: isCountTokens, - AddAnthropicVersion: true, - AnthropicVersion: DefaultVertexAnthropicVersion, - StripCacheControlScope: true, - RemapToolVersions: true, - InjectBetaHeadersIntoBody: true, - BetaHeaderOverrides: betaHeaderOverrides, - ProviderExtraHeaders: providerExtraHeaders, - ValidateTools: true, - ShouldSendBackRawRequest: shouldSendBackRawRequest, - ShouldSendBackRawResponse: shouldSendBackRawResponse, - }) - if buildErr != nil { - return nil, buildErr - } - stripped, err := anthropic.StripUnsupportedFieldsFromRawBody(jsonBody, schemas.Vertex, schemas.ResolveCanonicalModel(ctx, deployment)) - if err != nil { - return nil, providerUtils.NewBifrostOperationError(err.Error(), nil) - } - return stripped, nil -} - // isVertexMultiRegionEndpoint reports whether the Vertex location uses Google's // partner-model multi-region pool endpoint host instead of the single-region host. func isVertexMultiRegionEndpoint(region string) bool { diff --git a/core/providers/vertex/utils_test.go b/core/providers/vertex/utils_test.go index 3eee219c8f5..60dc344cc26 100644 --- a/core/providers/vertex/utils_test.go +++ b/core/providers/vertex/utils_test.go @@ -1,8 +1,10 @@ package vertex import ( + "reflect" "testing" + "github.com/maximhq/bifrost/core/providers/gemini" providerUtils "github.com/maximhq/bifrost/core/providers/utils" "github.com/maximhq/bifrost/core/schemas" ) @@ -54,6 +56,36 @@ func TestGetVertexAPIHost(t *testing.T) { } } +func TestVertexGeminiImageURLSchemesAllowGCS(t *testing.T) { + result, err := gemini.ToGeminiChatCompletionRequestWithImageURLSchemes(nil, &schemas.BifrostChatRequest{ + Model: "gemini-3-flash-preview", + Input: []schemas.ChatMessage{ + { + Role: schemas.ChatMessageRoleUser, + Content: &schemas.ChatMessageContent{ + ContentBlocks: []schemas.ChatContentBlock{ + { + Type: schemas.ChatContentBlockTypeImage, + ImageURLStruct: &schemas.ChatInputImage{ + URL: "gs://my-bucket/xxx.png", + }, + }, + }, + }, + }, + }, + }, geminiImageURLSchemes...) + if err != nil { + t.Fatalf("expected Vertex Gemini schemes to allow gs:// image URLs, got: %v", err) + } + if len(result.Contents) != 1 || len(result.Contents[0].Parts) != 1 || result.Contents[0].Parts[0].FileData == nil { + t.Fatalf("expected one fileData part, got %#v", result.Contents) + } + if result.Contents[0].Parts[0].FileData.FileURI != "gs://my-bucket/xxx.png" { + t.Fatalf("expected gs:// fileUri, got %q", result.Contents[0].Parts[0].FileData.FileURI) + } +} + func TestIsVertexMultiRegionEndpoint(t *testing.T) { t.Parallel() @@ -497,3 +529,15 @@ func TestResolveVertexProjectNumber_AliasOverride(t *testing.T) { t.Errorf("empty alias ProjectNumber should fall through: got %q, want %q", got, keyNumber) } } + +// TestGeminiImageURLSchemesContract pins the exact allowlist that Vertex passes +// into the Gemini converter. vertex.go reuses geminiImageURLSchemes across +// ChatCompletion / ChatCompletionStream / Responses / ResponsesStream / CountTokens, +// so a regression that drops "gs" (or accidentally adds e.g. "file") is caught +// here without having to drive each provider entrypoint end-to-end. +func TestGeminiImageURLSchemesContract(t *testing.T) { + want := []string{"http", "https", "gs"} + if !reflect.DeepEqual(geminiImageURLSchemes, want) { + t.Fatalf("geminiImageURLSchemes = %v, want %v", geminiImageURLSchemes, want) + } +} diff --git a/core/providers/vertex/vertex.go b/core/providers/vertex/vertex.go index cd107ed9dff..18b1e027f95 100644 --- a/core/providers/vertex/vertex.go +++ b/core/providers/vertex/vertex.go @@ -64,6 +64,12 @@ var vertexShortModelRe = regexp.MustCompile(`"(models/[^/"]+)"`) // is empty and we fall back to google.FindDefaultCredentials. const defaultCredentialsCacheKey = "__default_credentials__" +// geminiImageURLSchemes is the image URL scheme allowlist Vertex applies when it +// routes a request through the Gemini converter. Vertex natively accepts gs:// +// FileData URIs (in addition to http(s)), so we extend the Gemini-default list +// with "gs". +var geminiImageURLSchemes = []string{"http", "https", "gs"} + // getClientKey generates a unique key for caching token sources. // It uses a hash of the auth credentials for security. func getClientKey(authCredentials string) string { @@ -263,7 +269,7 @@ func (provider *VertexProvider) listModelsByKey(ctx *schemas.BifrostContext, key providerUtils.SetExtraHeaders(ctx, req, provider.networkConfig.ExtraHeaders, nil) req.Header.Set("Authorization", "Bearer "+token.AccessToken) - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) if bifrostErr != nil { wait() respBody := append([]byte(nil), resp.Body()...) @@ -301,9 +307,9 @@ func (provider *VertexProvider) listModelsByKey(ctx *schemas.BifrostContext, key var errorResp VertexError if err := sonic.Unmarshal(respBody, &errorResp); err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseUnmarshal, err), nil, respBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseUnmarshal, err), nil, respBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } - return nil, providerUtils.EnrichError(ctx, providerUtils.NewProviderAPIError(errorResp.Error.Message, nil, statusCode, nil, nil), nil, respBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewProviderAPIError(errorResp.Error.Message, nil, statusCode, nil, nil), nil, respBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Parse Vertex's publisher models response @@ -396,6 +402,11 @@ func inlineRemoteURLSources(ctx context.Context, request *schemas.BifrostChatReq if request == nil || request.Input == nil { return nil } + // When the caller is bypassing the converter via a pre-built raw body, + // the request struct isn't what gets sent — skip the fetch. + if useRawBody, ok := ctx.Value(schemas.BifrostContextKeyUseRawRequestBody).(bool); ok && useRawBody { + return nil + } for mi := range request.Input { msg := &request.Input[mi] if msg.Content == nil || msg.Content.ContentBlocks == nil { @@ -442,98 +453,131 @@ func inlineRemoteURLSources(ctx context.Context, request *schemas.BifrostChatReq return nil } -// ChatCompletion performs a chat completion request to the Vertex API. -// It supports both text and image content in messages. -// Returns a BifrostResponse containing the completion results or an error if the request fails. -func (provider *VertexProvider) ChatCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostChatRequest) (*schemas.BifrostChatResponse, *schemas.BifrostError) { - jsonBody, bifrostErr := providerUtils.CheckContextAndGetRequestBody( - ctx, - request, - func() (providerUtils.RequestBodyWithExtraParams, error) { - // Format messages for Vertex API, preserving key order for prompt caching - var rawBody []byte - var extraParams map[string]interface{} - var err error +// inlineDocumentURLsResponses is the Responses-API analogue of inlineDocumentURLs. +// File blocks live on ResponsesMessageContentBlock.ResponsesInputMessageContentBlockFile +// rather than the chat ContentBlock.File, so this walks the responses-shape input. +func inlineDocumentURLsResponses(ctx *schemas.BifrostContext, request *schemas.BifrostResponsesRequest) error { + if request == nil || request.Input == nil { + return nil + } + if useRawBody, ok := ctx.Value(schemas.BifrostContextKeyUseRawRequestBody).(bool); ok && useRawBody { + return nil + } + for mi := range request.Input { + msg := &request.Input[mi] + if msg.Content == nil || msg.Content.ContentBlocks == nil { + continue + } + for bi := range msg.Content.ContentBlocks { + block := &msg.Content.ContentBlocks[bi] - if schemas.IsAnthropicModelFamily(ctx, request.Model) { - // Anthropic-on-Vertex doesn't accept URL-source document or image blocks. - // Inline any URL documents/images to base64 before the converter runs. - if err := inlineRemoteURLSources(ctx, request); err != nil { - return nil, fmt.Errorf("failed to inline remote URL sources for vertex/claude: %w", err) - } - // Use centralized Anthropic converter - reqBody, convErr := anthropic.ToAnthropicChatRequest(ctx, request) - if convErr != nil { - return nil, convErr + // Inline url-source files. + if f := block.ResponsesInputMessageContentBlockFile; f != nil && f.FileURL != nil && *f.FileURL != "" { + mediaType, encoded, err := providerUtils.FetchAndEncodeURL(ctx, *f.FileURL) + if err != nil { + return err } - if reqBody == nil { - return nil, fmt.Errorf("chat completion input is not provided") + f.FileData = &encoded + if mediaType != "" && f.FileType == nil { + f.FileType = &mediaType } - extraParams = reqBody.GetExtraParams() - // Add provider-aware beta headers for Vertex - anthropic.AddMissingBetaHeadersToContext(ctx, reqBody, schemas.Vertex) - // Marshal to JSON bytes, preserving struct field order - rawBody, err = providerUtils.MarshalSorted(reqBody) + f.FileURL = nil + } + + // Inline url-source images to a base64 data URI; Anthropic-on-Vertex + // accepts base64 image sources only. Skip data: URIs (already inline). + if img := block.ResponsesInputMessageContentBlockImage; img != nil && img.ImageURL != nil && *img.ImageURL != "" && !strings.HasPrefix(*img.ImageURL, "data:") { + mediaType, encoded, err := providerUtils.FetchAndEncodeURL(ctx, *img.ImageURL) if err != nil { - return nil, fmt.Errorf("failed to marshal request body: %w", err) + return err } - // Add anthropic_version if not present (using sjson to preserve order) - if !providerUtils.JSONFieldExists(rawBody, "anthropic_version") { - rawBody, err = providerUtils.SetJSONField(rawBody, "anthropic_version", DefaultVertexAnthropicVersion) - if err != nil { - return nil, fmt.Errorf("failed to set anthropic_version: %w", err) + if mediaType != "" { + dataURI := "data:" + mediaType + ";base64," + encoded + img.ImageURL = &dataURI + } else { + // Content-Type header absent; sniff the media type from the + // fetched bytes so we never emit a malformed "data:;base64,..." + // URI, which Anthropic-on-Vertex rejects. + sanitized, sErr := schemas.SanitizeImageURL(encoded) + if sErr != nil { + return sErr } + img.ImageURL = &sanitized } - // Inject beta headers into body as anthropic_beta (Vertex uses body field, not HTTP header) - if betaHeaders := anthropic.FilterBetaHeadersForProvider(anthropic.MergeBetaHeaders(ctx, provider.networkConfig.ExtraHeaders), schemas.Vertex, provider.networkConfig.BetaHeaderOverrides); len(betaHeaders) > 0 { - rawBody, err = providerUtils.SetJSONField(rawBody, "anthropic_beta", betaHeaders) + } + } + } + return nil +} + +// ChatCompletion performs a chat completion request to the Vertex API. +// It supports both text and image content in messages. +// Returns a BifrostResponse containing the completion results or an error if the request fails. +func (provider *VertexProvider) ChatCompletion(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostChatRequest) (*schemas.BifrostChatResponse, *schemas.BifrostError) { + var jsonBody []byte + var bifrostErr *schemas.BifrostError + if schemas.IsAnthropicModelFamily(ctx, request.Model) { + // Anthropic-on-Vertex doesn't accept URL-source document or image blocks. + // Inline any URL documents/images to base64 before the converter runs. + if err := inlineRemoteURLSources(ctx, request); err != nil { + return nil, providerUtils.NewBifrostOperationError("failed to inline remote URL sources for vertex/claude", err) + } + jsonBody, bifrostErr = anthropic.BuildAnthropicChatRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{ + Provider: schemas.Vertex, + Model: request.Model, + BetaHeaderOverrides: provider.networkConfig.BetaHeaderOverrides, + ProviderExtraHeaders: provider.networkConfig.ExtraHeaders, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }) + } else { + jsonBody, bifrostErr = providerUtils.CheckContextAndGetRequestBody( + ctx, + request, + func() (providerUtils.RequestBodyWithExtraParams, error) { + // Format messages for Vertex API, preserving key order for prompt caching + var rawBody []byte + var extraParams map[string]interface{} + var err error + + if schemas.IsGeminiModelFamily(ctx, request.Model) || schemas.IsAllDigitsASCII(request.Model) || schemas.IsGemmaModelFamily(ctx, request.Model) { + reqBody, err := gemini.ToGeminiChatCompletionRequestWithImageURLSchemes(ctx, request, geminiImageURLSchemes...) if err != nil { - return nil, fmt.Errorf("failed to set anthropic_beta: %w", err) + return nil, err + } + if reqBody == nil { + return nil, fmt.Errorf("chat completion input is not provided") + } + extraParams = reqBody.GetExtraParams() + // Strip unsupported fields for Vertex Gemini + stripVertexGeminiUnsupportedFields(reqBody) + // Marshal to JSON bytes + rawBody, err = providerUtils.MarshalSorted(reqBody) + if err != nil { + return nil, fmt.Errorf("failed to marshal request body: %w", err) + } + } else { + // Use centralized OpenAI converter for non-Claude models + reqBody := openai.ToOpenAIChatRequest(ctx, request) + if reqBody == nil { + return nil, fmt.Errorf("chat completion input is not provided") + } + extraParams = reqBody.GetExtraParams() + // Marshal to JSON bytes + rawBody, err = providerUtils.MarshalSorted(reqBody) + if err != nil { + return nil, fmt.Errorf("failed to marshal request body: %w", err) } } - // Remove model field (it's in URL for Vertex) - rawBody, err = providerUtils.DeleteJSONField(rawBody, "model") - if err != nil { - return nil, fmt.Errorf("failed to delete model field: %w", err) - } - } else if schemas.IsGeminiModelFamily(ctx, request.Model) || schemas.IsAllDigitsASCII(request.Model) || schemas.IsGemmaModelFamily(ctx, request.Model) { - reqBody, err := gemini.ToGeminiChatCompletionRequest(ctx, request) - if err != nil { - return nil, err - } - if reqBody == nil { - return nil, fmt.Errorf("chat completion input is not provided") - } - extraParams = reqBody.GetExtraParams() - // Strip unsupported fields for Vertex Gemini - stripVertexGeminiUnsupportedFields(reqBody) - // Marshal to JSON bytes - rawBody, err = providerUtils.MarshalSorted(reqBody) - if err != nil { - return nil, fmt.Errorf("failed to marshal request body: %w", err) - } - } else { - // Use centralized OpenAI converter for non-Claude models - reqBody := openai.ToOpenAIChatRequest(ctx, request) - if reqBody == nil { - return nil, fmt.Errorf("chat completion input is not provided") - } - extraParams = reqBody.GetExtraParams() - // Marshal to JSON bytes - rawBody, err = providerUtils.MarshalSorted(reqBody) + // Remove region field if present + rawBody, err = providerUtils.DeleteJSONField(rawBody, "region") if err != nil { - return nil, fmt.Errorf("failed to marshal request body: %w", err) + return nil, fmt.Errorf("failed to delete region field: %w", err) } - } - - // Remove region field if present - rawBody, err = providerUtils.DeleteJSONField(rawBody, "region") - if err != nil { - return nil, fmt.Errorf("failed to delete region field: %w", err) - } - return &VertexRawRequestBody{RawBody: rawBody, ExtraParams: extraParams}, nil - }, - ) + return &VertexRawRequestBody{RawBody: rawBody, ExtraParams: extraParams}, nil + }, + ) + } if bifrostErr != nil { return nil, bifrostErr } @@ -652,7 +696,7 @@ func (provider *VertexProvider) ChatCompletion(ctx *schemas.BifrostContext, key latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if usedLargePayloadBody { providerUtils.DrainLargePayloadRemainder(ctx) @@ -665,12 +709,12 @@ func (provider *VertexProvider) ChatCompletion(ctx *schemas.BifrostContext, key if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } responseBody, isLargeResp, decodeErr := providerUtils.FinalizeResponseWithLargeDetection(ctx, resp, provider.logger) if decodeErr != nil { - return nil, providerUtils.EnrichError(ctx, decodeErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, decodeErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if isLargeResp { respOwned = false @@ -690,7 +734,7 @@ func (provider *VertexProvider) ChatCompletion(ctx *schemas.BifrostContext, key rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, anthropicResponse, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Create final response @@ -717,7 +761,7 @@ func (provider *VertexProvider) ChatCompletion(ctx *schemas.BifrostContext, key rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, &geminiResponse, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } response := geminiResponse.ToBifrostChatResponse() @@ -739,7 +783,7 @@ func (provider *VertexProvider) ChatCompletion(ctx *schemas.BifrostContext, key // Use enhanced response handler with pre-allocated response rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, response, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } response.ExtraFields.Latency = latency.Milliseconds() @@ -775,62 +819,20 @@ func (provider *VertexProvider) ChatCompletionStream(ctx *schemas.BifrostContext } if schemas.IsAnthropicModelFamily(ctx, request.Model) { - // Use Anthropic-style streaming for Claude models - jsonData, bifrostErr := providerUtils.CheckContextAndGetRequestBody( - ctx, - request, - func() (providerUtils.RequestBodyWithExtraParams, error) { - var extraParams map[string]interface{} - // Anthropic-on-Vertex doesn't accept URL-source document or image blocks. - // Inline any URL documents/images to base64 before the converter runs. - if err := inlineRemoteURLSources(ctx, request); err != nil { - return nil, fmt.Errorf("failed to inline remote URL sources for vertex/claude: %w", err) - } - reqBody, convErr := anthropic.ToAnthropicChatRequest(ctx, request) - if convErr != nil { - return nil, convErr - } - if reqBody == nil { - return nil, fmt.Errorf("chat completion input is not provided") - } - extraParams = reqBody.GetExtraParams() - reqBody.Stream = new(true) - // Add provider-aware beta headers for Vertex - anthropic.AddMissingBetaHeadersToContext(ctx, reqBody, schemas.Vertex) - - // Marshal to JSON bytes, preserving struct field order for prompt caching - rawBody, err := providerUtils.MarshalSorted(reqBody) - if err != nil { - return nil, fmt.Errorf("failed to marshal request body: %w", err) - } - - // Add anthropic_version if not present (using sjson to preserve order) - if !providerUtils.JSONFieldExists(rawBody, "anthropic_version") { - rawBody, err = providerUtils.SetJSONField(rawBody, "anthropic_version", DefaultVertexAnthropicVersion) - if err != nil { - return nil, fmt.Errorf("failed to set anthropic_version: %w", err) - } - } - // Inject beta headers into body as anthropic_beta (Vertex uses body field, not HTTP header) - if betaHeaders := anthropic.FilterBetaHeadersForProvider(anthropic.MergeBetaHeaders(ctx, provider.networkConfig.ExtraHeaders), schemas.Vertex, provider.networkConfig.BetaHeaderOverrides); len(betaHeaders) > 0 { - rawBody, err = providerUtils.SetJSONField(rawBody, "anthropic_beta", betaHeaders) - if err != nil { - return nil, fmt.Errorf("failed to set anthropic_beta: %w", err) - } - } - - // Remove model and region fields (using sjson to preserve order) - rawBody, err = providerUtils.DeleteJSONField(rawBody, "model") - if err != nil { - return nil, fmt.Errorf("failed to delete model field: %w", err) - } - rawBody, err = providerUtils.DeleteJSONField(rawBody, "region") - if err != nil { - return nil, fmt.Errorf("failed to delete region field: %w", err) - } - return &VertexRawRequestBody{RawBody: rawBody, ExtraParams: extraParams}, nil - }, - ) + // Use Anthropic-style streaming for Claude models. + // Anthropic-on-Vertex doesn't accept URL-source document or image blocks; inline first. + if err := inlineRemoteURLSources(ctx, request); err != nil { + return nil, providerUtils.NewBifrostOperationError("failed to inline remote URL sources for vertex/claude", err) + } + jsonData, bifrostErr := anthropic.BuildAnthropicChatRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{ + Provider: schemas.Vertex, + Model: request.Model, + IsStreaming: true, + BetaHeaderOverrides: provider.networkConfig.BetaHeaderOverrides, + ProviderExtraHeaders: provider.networkConfig.ExtraHeaders, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }) if bifrostErr != nil { return nil, bifrostErr } @@ -887,6 +889,7 @@ func (provider *VertexProvider) ChatCompletionStream(ctx *schemas.BifrostContext providerName, postHookRunner, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -896,7 +899,7 @@ func (provider *VertexProvider) ChatCompletionStream(ctx *schemas.BifrostContext ctx, request, func() (providerUtils.RequestBodyWithExtraParams, error) { - reqBody, err := gemini.ToGeminiChatCompletionRequest(ctx, request) + reqBody, err := gemini.ToGeminiChatCompletionRequestWithImageURLSchemes(ctx, request, geminiImageURLSchemes...) if err != nil { return nil, err } @@ -971,6 +974,7 @@ func (provider *VertexProvider) ChatCompletionStream(ctx *schemas.BifrostContext jsonData, headers, provider.networkConfig.ExtraHeaders, + provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), @@ -1032,6 +1036,7 @@ func (provider *VertexProvider) ChatCompletionStream(ctx *schemas.BifrostContext nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -1041,7 +1046,20 @@ func (provider *VertexProvider) ChatCompletionStream(ctx *schemas.BifrostContext // Responses performs a responses request to the Vertex API. func (provider *VertexProvider) Responses(ctx *schemas.BifrostContext, key schemas.Key, request *schemas.BifrostResponsesRequest) (*schemas.BifrostResponsesResponse, *schemas.BifrostError) { if schemas.IsAnthropicModelFamily(ctx, request.Model) { - jsonBody, bifrostErr := getRequestBodyForAnthropicResponses(ctx, request, request.Model, false, false, provider.networkConfig.BetaHeaderOverrides, provider.networkConfig.ExtraHeaders, provider.sendBackRawRequest, provider.sendBackRawResponse) + // Anthropic-on-Vertex doesn't accept URL-source document blocks. + // Inline any URL documents to base64 before the converter runs. + if err := inlineDocumentURLsResponses(ctx, request); err != nil { + return nil, providerUtils.NewBifrostOperationError("failed to inline document URLs for vertex/claude", err) + } + jsonBody, bifrostErr := anthropic.BuildAnthropicResponsesRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{ + Provider: schemas.Vertex, + Model: request.Model, + BetaHeaderOverrides: provider.networkConfig.BetaHeaderOverrides, + ProviderExtraHeaders: provider.networkConfig.ExtraHeaders, + ValidateTools: true, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }) if bifrostErr != nil { return nil, bifrostErr } @@ -1101,7 +1119,7 @@ func (provider *VertexProvider) Responses(ctx *schemas.BifrostContext, key schem latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if usedLargePayloadBody { providerUtils.DrainLargePayloadRemainder(ctx) @@ -1114,12 +1132,12 @@ func (provider *VertexProvider) Responses(ctx *schemas.BifrostContext, key schem if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } responseBody, isLargeResp, decodeErr := providerUtils.FinalizeResponseWithLargeDetection(ctx, resp, provider.logger) if decodeErr != nil { - return nil, providerUtils.EnrichError(ctx, decodeErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, decodeErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if isLargeResp { respOwned = false @@ -1137,7 +1155,7 @@ func (provider *VertexProvider) Responses(ctx *schemas.BifrostContext, key schem rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, anthropicResponse, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Create final response @@ -1164,7 +1182,7 @@ func (provider *VertexProvider) Responses(ctx *schemas.BifrostContext, key schem ctx, request, func() (providerUtils.RequestBodyWithExtraParams, error) { - reqBody, err := gemini.ToGeminiResponsesRequest(ctx, request) + reqBody, err := gemini.ToGeminiResponsesRequestWithImageURLSchemes(ctx, request, geminiImageURLSchemes...) if err != nil { return nil, err } @@ -1257,7 +1275,7 @@ func (provider *VertexProvider) Responses(ctx *schemas.BifrostContext, key schem latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if usedLargePayloadBody { providerUtils.DrainLargePayloadRemainder(ctx) @@ -1270,12 +1288,12 @@ func (provider *VertexProvider) Responses(ctx *schemas.BifrostContext, key schem if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } responseBody, isLargeResp, decodeErr := providerUtils.FinalizeResponseWithLargeDetection(ctx, resp, provider.logger) if decodeErr != nil { - return nil, providerUtils.EnrichError(ctx, decodeErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, decodeErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if isLargeResp { respOwned = false @@ -1291,7 +1309,7 @@ func (provider *VertexProvider) Responses(ctx *schemas.BifrostContext, key schem rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, geminiResponse, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } response := geminiResponse.ToResponsesBifrostResponsesResponse() @@ -1332,7 +1350,21 @@ func (provider *VertexProvider) ResponsesStream(ctx *schemas.BifrostContext, pos return nil, providerUtils.NewConfigurationError("project ID is not set") } - jsonBody, bifrostErr := getRequestBodyForAnthropicResponses(ctx, request, request.Model, true, false, provider.networkConfig.BetaHeaderOverrides, provider.networkConfig.ExtraHeaders, provider.sendBackRawRequest, provider.sendBackRawResponse) + // Anthropic-on-Vertex doesn't accept URL-source document blocks. + // Inline any URL documents to base64 before the converter runs. + if err := inlineDocumentURLsResponses(ctx, request); err != nil { + return nil, providerUtils.NewBifrostOperationError("failed to inline document URLs for vertex/claude", err) + } + jsonBody, bifrostErr := anthropic.BuildAnthropicResponsesRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{ + Provider: schemas.Vertex, + Model: request.Model, + IsStreaming: true, + BetaHeaderOverrides: provider.networkConfig.BetaHeaderOverrides, + ProviderExtraHeaders: provider.networkConfig.ExtraHeaders, + ValidateTools: true, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }) if bifrostErr != nil { return nil, bifrostErr } @@ -1372,6 +1404,7 @@ func (provider *VertexProvider) ResponsesStream(ctx *schemas.BifrostContext, pos provider.GetProviderKey(), postHookRunner, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -1391,7 +1424,7 @@ func (provider *VertexProvider) ResponsesStream(ctx *schemas.BifrostContext, pos ctx, request, func() (providerUtils.RequestBodyWithExtraParams, error) { - reqBody, err := gemini.ToGeminiResponsesRequest(ctx, request) + reqBody, err := gemini.ToGeminiResponsesRequestWithImageURLSchemes(ctx, request, geminiImageURLSchemes...) if err != nil { return nil, err } @@ -1469,6 +1502,7 @@ func (provider *VertexProvider) ResponsesStream(ctx *schemas.BifrostContext, pos jsonData, headers, provider.networkConfig.ExtraHeaders, + provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), @@ -1573,7 +1607,7 @@ func (provider *VertexProvider) Embedding(ctx *schemas.BifrostContext, key schem latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if usedLargePayloadBody { providerUtils.DrainLargePayloadRemainder(ctx) @@ -1595,7 +1629,7 @@ func (provider *VertexProvider) Embedding(ctx *schemas.BifrostContext, key schem // Try to parse Vertex's error format var vertexError map[string]interface{} if err := sonic.Unmarshal(errBody, &vertexError); err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseUnmarshal, err), jsonBody, errBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError(schemas.ErrProviderResponseUnmarshal, err), jsonBody, errBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if errorObj, exists := vertexError["error"]; exists { @@ -1609,7 +1643,7 @@ func (provider *VertexProvider) Embedding(ctx *schemas.BifrostContext, key schem } } - return nil, providerUtils.EnrichError(ctx, providerUtils.NewProviderAPIError(errorMessage, nil, resp.StatusCode(), nil, nil), jsonBody, errBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewProviderAPIError(errorMessage, nil, resp.StatusCode(), nil, nil), jsonBody, errBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } responseBody, isLargeResp, decodeErr := providerUtils.FinalizeResponseWithLargeDetection(ctx, resp, provider.logger) @@ -1718,7 +1752,7 @@ func (provider *VertexProvider) Rerank(ctx *schemas.BifrostContext, key schemas. latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if usedLargePayloadBody { providerUtils.DrainLargePayloadRemainder(ctx) @@ -1746,12 +1780,12 @@ func (provider *VertexProvider) Rerank(ctx *schemas.BifrostContext, key schemas. } } - return nil, providerUtils.EnrichError(ctx, parsedError, jsonBody, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parsedError, jsonBody, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } responseBody, isLargeResp, decodeErr := providerUtils.FinalizeResponseWithLargeDetection(ctx, resp, provider.logger) if decodeErr != nil { - return nil, providerUtils.EnrichError(ctx, decodeErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, decodeErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if isLargeResp { respOwned = false @@ -1767,13 +1801,13 @@ func (provider *VertexProvider) Rerank(ctx *schemas.BifrostContext, key schemas. vertexResponse := &VertexRankResponse{} rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, vertexResponse, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } returnDocuments := request.Params != nil && request.Params.ReturnDocuments != nil && *request.Params.ReturnDocuments bifrostResponse, err := vertexResponse.ToBifrostRerankResponse(request.Documents, returnDocuments) if err != nil { - return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("error converting rerank response", err), jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, providerUtils.NewBifrostOperationError("error converting rerank response", err), jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } bifrostResponse.ExtraFields.Latency = latency.Milliseconds() @@ -1943,7 +1977,7 @@ func (provider *VertexProvider) ImageGeneration(ctx *schemas.BifrostContext, key latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if usedLargePayloadBody { providerUtils.DrainLargePayloadRemainder(ctx) @@ -1956,12 +1990,12 @@ func (provider *VertexProvider) ImageGeneration(ctx *schemas.BifrostContext, key if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } responseBody, isLargeResp, decodeErr := providerUtils.FinalizeResponseWithLargeDetection(ctx, resp, provider.logger) if decodeErr != nil { - return nil, providerUtils.EnrichError(ctx, decodeErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, decodeErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if isLargeResp { respOwned = false @@ -1978,12 +2012,12 @@ func (provider *VertexProvider) ImageGeneration(ctx *schemas.BifrostContext, key rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, &geminiResponse, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } response, err := geminiResponse.ToBifrostImageGenerationResponse() if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } response.ExtraFields.Latency = latency.Milliseconds() @@ -2004,7 +2038,7 @@ func (provider *VertexProvider) ImageGeneration(ctx *schemas.BifrostContext, key rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, &imagenResponse, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } response := imagenResponse.ToBifrostImageGenerationResponse() @@ -2152,7 +2186,7 @@ func (provider *VertexProvider) ImageEdit(ctx *schemas.BifrostContext, key schem latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, activeClient, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if usedLargePayloadBody { providerUtils.DrainLargePayloadRemainder(ctx) @@ -2164,12 +2198,12 @@ func (provider *VertexProvider) ImageEdit(ctx *schemas.BifrostContext, key schem if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } responseBody, isLargeResp, decodeErr := providerUtils.FinalizeResponseWithLargeDetection(ctx, resp, provider.logger) if decodeErr != nil { - return nil, providerUtils.EnrichError(ctx, decodeErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, decodeErr, jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if isLargeResp { respOwned = false @@ -2186,12 +2220,12 @@ func (provider *VertexProvider) ImageEdit(ctx *schemas.BifrostContext, key schem rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, &geminiResponse, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } response, err := geminiResponse.ToBifrostImageGenerationResponse() if err != nil { - return nil, providerUtils.EnrichError(ctx, err, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, err, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } response.ExtraFields.Latency = latency.Milliseconds() @@ -2212,7 +2246,7 @@ func (provider *VertexProvider) ImageEdit(ctx *schemas.BifrostContext, key schem rawRequest, rawResponse, bifrostErr := providerUtils.HandleProviderResponse(responseBody, &imagenResponse, jsonBody, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, responseBody, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } response := imagenResponse.ToBifrostImageGenerationResponse() @@ -2320,7 +2354,7 @@ func (provider *VertexProvider) VideoGeneration(ctx *schemas.BifrostContext, key latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerUtils.ExtractProviderResponseHeaders(resp)) @@ -2329,7 +2363,7 @@ func (provider *VertexProvider) VideoGeneration(ctx *schemas.BifrostContext, key if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // Parse response @@ -2436,7 +2470,7 @@ func (provider *VertexProvider) VideoRetrieve(ctx *schemas.BifrostContext, key s latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerUtils.ExtractProviderResponseHeaders(resp)) @@ -2445,7 +2479,7 @@ func (provider *VertexProvider) VideoRetrieve(ctx *schemas.BifrostContext, key s if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonBody, nil, sendBackRawRequest, sendBackRawResponse, latency) } // Parse response @@ -2551,9 +2585,9 @@ func (provider *VertexProvider) VideoDownload(ctx *schemas.BifrostContext, key s } ctx.SetValue(schemas.BifrostContextKeyProviderResponseHeaders, providerUtils.ExtractProviderResponseHeaders(resp)) if resp.StatusCode() != fasthttp.StatusOK { - return nil, providerUtils.NewBifrostOperationError( + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostOperationError( fmt.Sprintf("failed to download video: HTTP %d", resp.StatusCode()), - nil) + nil), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) if err != nil { @@ -2779,23 +2813,23 @@ func (provider *VertexProvider) BatchCreate(ctx *schemas.BifrostContext, key sch sendBackRawResponse := providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) startTime := time.Now() - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if resp.StatusCode() != fasthttp.StatusOK { if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, providerUtils.EnrichError(ctx, parseVertexJobAPIError(resp.Body(), resp.StatusCode(), "batch create"), jsonData, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseVertexJobAPIError(resp.Body(), resp.StatusCode(), "batch create"), jsonData, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } var created VertexBatchPredictionJob rawRequest, rawResponse, parseErr := providerUtils.HandleProviderResponse(resp.Body(), &created, jsonData, sendBackRawRequest, sendBackRawResponse) if parseErr != nil { - return nil, providerUtils.EnrichError(ctx, parseErr, jsonData, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseErr, jsonData, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // In raw-passthrough mode inputFileID may be empty (e.g. BigQuery or multi-URI inputs the @@ -2920,24 +2954,24 @@ func (provider *VertexProvider) batchListByKey(ctx *schemas.BifrostContext, key sendBackRawResponse := providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse) startTime := time.Now() - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { - return nil, 0, providerUtils.EnrichError(ctx, bifrostErr, nil, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, 0, providerUtils.EnrichError(ctx, bifrostErr, nil, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if resp.StatusCode() != fasthttp.StatusOK { if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, 0, providerUtils.EnrichError(ctx, parseVertexJobAPIError(resp.Body(), resp.StatusCode(), "batch list"), nil, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, 0, providerUtils.EnrichError(ctx, parseVertexJobAPIError(resp.Body(), resp.StatusCode(), "batch list"), nil, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } // GET request: no request body, so raw request capture is skipped by HandleProviderResponse. var listResp VertexBatchJobListResponse _, rawResponse, parseErr := providerUtils.HandleProviderResponse(resp.Body(), &listResp, nil, false, sendBackRawResponse) if parseErr != nil { - return nil, 0, providerUtils.EnrichError(ctx, parseErr, nil, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, 0, providerUtils.EnrichError(ctx, parseErr, nil, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } data := make([]schemas.BifrostBatchRetrieveResponse, 0, len(listResp.BatchPredictionJobs)) @@ -3016,23 +3050,23 @@ func (provider *VertexProvider) vertexGetBatchJob(ctx *schemas.BifrostContext, k providerUtils.SetExtraHeaders(ctx, req, provider.networkConfig.ExtraHeaders, nil) req.Header.Set("Authorization", authHeader) - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { - return nil, nil, providerUtils.EnrichError(ctx, bifrostErr, nil, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, nil, providerUtils.EnrichError(ctx, bifrostErr, nil, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if resp.StatusCode() != fasthttp.StatusOK { if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, nil, providerUtils.EnrichError(ctx, parseVertexJobAPIError(resp.Body(), resp.StatusCode(), "batch retrieve"), nil, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, nil, providerUtils.EnrichError(ctx, parseVertexJobAPIError(resp.Body(), resp.StatusCode(), "batch retrieve"), nil, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } var job VertexBatchPredictionJob _, rawResponse, parseErr := providerUtils.HandleProviderResponse(resp.Body(), &job, nil, false, providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)) if parseErr != nil { - return nil, nil, providerUtils.EnrichError(ctx, parseErr, nil, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, nil, providerUtils.EnrichError(ctx, parseErr, nil, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } return &job, rawResponse, nil } @@ -3080,17 +3114,17 @@ func (provider *VertexProvider) batchCancelByKey(ctx *schemas.BifrostContext, ke req.Header.Set("Authorization", authHeader) startTime := time.Now() - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, nil, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, nil, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if resp.StatusCode() != fasthttp.StatusOK { if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, providerUtils.EnrichError(ctx, parseVertexJobAPIError(resp.Body(), resp.StatusCode(), "batch cancel"), nil, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseVertexJobAPIError(resp.Body(), resp.StatusCode(), "batch cancel"), nil, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } return &schemas.BifrostBatchCancelResponse{ @@ -3147,17 +3181,17 @@ func (provider *VertexProvider) batchDeleteByKey(ctx *schemas.BifrostContext, ke req.Header.Set("Authorization", authHeader) startTime := time.Now() - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, nil, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, nil, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } if resp.StatusCode() != fasthttp.StatusOK { if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, providerUtils.EnrichError(ctx, parseVertexJobAPIError(resp.Body(), resp.StatusCode(), "batch delete"), nil, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseVertexJobAPIError(resp.Body(), resp.StatusCode(), "batch delete"), nil, resp.Body(), provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } return &schemas.BifrostBatchDeleteResponse{ @@ -3334,14 +3368,14 @@ func (provider *VertexProvider) gcsDownloadObject(ctx *schemas.BifrostContext, a providerUtils.SetExtraHeaders(ctx, req, provider.networkConfig.ExtraHeaders, nil) req.Header.Set("Authorization", authHeader) - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { return nil, bifrostErr } if resp.StatusCode() != fasthttp.StatusOK { - return nil, parseGCSAPIError(resp.Body(), resp.StatusCode(), "content download") + return nil, providerUtils.SetErrorLatency(parseGCSAPIError(resp.Body(), resp.StatusCode(), "content download"), latency) } content := make([]byte, len(resp.Body())) @@ -3551,7 +3585,7 @@ func (provider *VertexProvider) gcsFileUploadDirect( req.Header.Set("Authorization", authHeader) req.SetBody(buf.Bytes()) - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { return nil, bifrostErr @@ -3561,7 +3595,7 @@ func (provider *VertexProvider) gcsFileUploadDirect( if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, parseGCSAPIError(resp.Body(), resp.StatusCode(), "upload") + return nil, providerUtils.SetErrorLatency(parseGCSAPIError(resp.Body(), resp.StatusCode(), "upload"), latency) } return &schemas.BifrostFileUploadResponse{ @@ -3625,7 +3659,7 @@ func (provider *VertexProvider) gcsFileUploadResumable( } } - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { return nil, bifrostErr @@ -3635,7 +3669,7 @@ func (provider *VertexProvider) gcsFileUploadResumable( if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, parseGCSAPIError(resp.Body(), resp.StatusCode(), "resumable session initiation") + return nil, providerUtils.SetErrorLatency(parseGCSAPIError(resp.Body(), resp.StatusCode(), "resumable session initiation"), latency) } sessionURL := string(resp.Header.Peek("Location")) @@ -3723,7 +3757,7 @@ func (provider *VertexProvider) FileList(ctx *schemas.BifrostContext, keys []sch req.Header.Set("Authorization", authHeader) startTime := time.Now() - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { return nil, bifrostErr @@ -3733,7 +3767,7 @@ func (provider *VertexProvider) FileList(ctx *schemas.BifrostContext, keys []sch if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, parseGCSAPIError(resp.Body(), resp.StatusCode(), "list") + return nil, providerUtils.SetErrorLatency(parseGCSAPIError(resp.Body(), resp.StatusCode(), "list"), latency) } var listResp gcsObjectListResponse @@ -3806,7 +3840,7 @@ func (provider *VertexProvider) fileRetrieveByKey(ctx *schemas.BifrostContext, k req.Header.Set("Authorization", authHeader) startTime := time.Now() - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { return nil, bifrostErr @@ -3816,7 +3850,7 @@ func (provider *VertexProvider) fileRetrieveByKey(ctx *schemas.BifrostContext, k if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, parseGCSAPIError(resp.Body(), resp.StatusCode(), "retrieve") + return nil, providerUtils.SetErrorLatency(parseGCSAPIError(resp.Body(), resp.StatusCode(), "retrieve"), latency) } var obj gcsObjectMetadata @@ -3893,7 +3927,7 @@ func (provider *VertexProvider) fileDeleteByKey(ctx *schemas.BifrostContext, key req.Header.Set("Authorization", authHeader) startTime := time.Now() - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { return nil, bifrostErr @@ -3904,7 +3938,7 @@ func (provider *VertexProvider) fileDeleteByKey(ctx *schemas.BifrostContext, key if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, parseGCSAPIError(resp.Body(), resp.StatusCode(), "delete") + return nil, providerUtils.SetErrorLatency(parseGCSAPIError(resp.Body(), resp.StatusCode(), "delete"), latency) } return &schemas.BifrostFileDeleteResponse{ @@ -3959,7 +3993,7 @@ func (provider *VertexProvider) fileContentByKey(ctx *schemas.BifrostContext, ke req.Header.Set("Authorization", authHeader) startTime := time.Now() - _, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) + latency, bifrostErr, wait := providerUtils.MakeRequestWithContext(ctx, provider.client, req, resp) defer wait() if bifrostErr != nil { return nil, bifrostErr @@ -3969,7 +4003,7 @@ func (provider *VertexProvider) fileContentByKey(ctx *schemas.BifrostContext, ke if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, parseGCSAPIError(resp.Body(), resp.StatusCode(), "content download") + return nil, providerUtils.SetErrorLatency(parseGCSAPIError(resp.Body(), resp.StatusCode(), "content download"), latency) } // Copy body before deferred ReleaseResponse invalidates the buffer. @@ -4000,7 +4034,21 @@ func (provider *VertexProvider) CountTokens(ctx *schemas.BifrostContext, key sch ) if schemas.IsAnthropicModelFamily(ctx, request.Model) { - jsonBody, bifrostErr = getRequestBodyForAnthropicResponses(ctx, request, request.Model, false, true, provider.networkConfig.BetaHeaderOverrides, provider.networkConfig.ExtraHeaders, provider.sendBackRawRequest, provider.sendBackRawResponse) + // Anthropic-on-Vertex doesn't accept URL-source document blocks. + // Inline any URL documents to base64 before the converter runs. + if err := inlineDocumentURLsResponses(ctx, request); err != nil { + return nil, providerUtils.NewBifrostOperationError("failed to inline document URLs for vertex/claude", err) + } + jsonBody, bifrostErr = anthropic.BuildAnthropicResponsesRequestBody(ctx, request, anthropic.AnthropicRequestBuildConfig{ + Provider: schemas.Vertex, + Model: request.Model, + IsCountTokens: true, + BetaHeaderOverrides: provider.networkConfig.BetaHeaderOverrides, + ProviderExtraHeaders: provider.networkConfig.ExtraHeaders, + ValidateTools: true, + ShouldSendBackRawRequest: provider.sendBackRawRequest, + ShouldSendBackRawResponse: provider.sendBackRawResponse, + }) if bifrostErr != nil { return nil, bifrostErr } @@ -4009,7 +4057,7 @@ func (provider *VertexProvider) CountTokens(ctx *schemas.BifrostContext, key sch ctx, request, func() (providerUtils.RequestBodyWithExtraParams, error) { - return gemini.ToGeminiResponsesRequest(ctx, request) + return gemini.ToGeminiResponsesRequestWithImageURLSchemes(ctx, request, geminiImageURLSchemes...) }, ) if bifrostErr != nil { @@ -4116,7 +4164,7 @@ func (provider *VertexProvider) CountTokens(ctx *schemas.BifrostContext, key sch if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { removeVertexClient(key.VertexKeyConfig.AuthCredentials.GetValue()) } - return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, parseVertexError(resp), jsonBody, nil, provider.sendBackRawRequest, provider.sendBackRawResponse, latency) } responseBody, isLargeResp, decodeErr := providerUtils.FinalizeResponseWithLargeDetection(ctx, resp, provider.logger) @@ -4476,26 +4524,29 @@ func (provider *VertexProvider) PassthroughStream( } activeClient := providerUtils.PrepareResponseStreaming(ctx, provider.streamingClient, resp) - if err := activeClient.Do(fasthttpReq, resp); err != nil { + startTime := time.Now() + err := activeClient.Do(fasthttpReq, resp) + latency := time.Since(startTime) + if err != nil { providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } // Request failed before the first response byte (server closed an idle/pooled connection, // broken pipe, connection refused, DNS failure, etc.). Surface as a retriable upstream // connection error (502) so executeRequestWithRetries honors max_retries, matching the // non-streaming path - see https://github.com/maximhq/bifrost/issues/4496. - return nil, providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostUpstreamConnectionError(schemas.ErrProviderDoRequest, err), latency) } if resp.StatusCode() == fasthttp.StatusUnauthorized || resp.StatusCode() == fasthttp.StatusForbidden { diff --git a/core/providers/vertex/vertex_test.go b/core/providers/vertex/vertex_test.go index d29499f2329..02e0f271dd4 100644 --- a/core/providers/vertex/vertex_test.go +++ b/core/providers/vertex/vertex_test.go @@ -6,8 +6,11 @@ import ( "testing" "github.com/maximhq/bifrost/core/internal/llmtests" + "github.com/maximhq/bifrost/core/providers/gemini" "github.com/maximhq/bifrost/core/schemas" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" ) func TestVertex(t *testing.T) { @@ -113,3 +116,95 @@ func TestVertex(t *testing.T) { llmtests.RunAllComprehensiveTests(t, client, ctx, testConfig) }) } + +func TestVertexGeminiChatCompletionPreservesGCSImageURL(t *testing.T) { + result, err := gemini.ToGeminiChatCompletionRequestWithImageURLSchemes(nil, &schemas.BifrostChatRequest{ + Model: "gemini-3-flash-preview", + Input: []schemas.ChatMessage{ + { + Role: schemas.ChatMessageRoleUser, + Content: &schemas.ChatMessageContent{ + ContentBlocks: []schemas.ChatContentBlock{ + { + Type: schemas.ChatContentBlockTypeText, + Text: schemas.Ptr("Describe this image."), + }, + { + Type: schemas.ChatContentBlockTypeImage, + ImageURLStruct: &schemas.ChatInputImage{ + URL: "gs://my-bucket/xxx.png", + }, + }, + }, + }, + }, + }, + }, "http", "https", "gs") + + require.NoError(t, err) + require.Len(t, result.Contents, 1) + require.Len(t, result.Contents[0].Parts, 2) + require.NotNil(t, result.Contents[0].Parts[1].FileData) + assert.Equal(t, "gs://my-bucket/xxx.png", result.Contents[0].Parts[1].FileData.FileURI) + assert.Equal(t, "image/png", result.Contents[0].Parts[1].FileData.MIMEType) +} + +func TestVertexGeminiResponsesPreservesGCSImageURL(t *testing.T) { + result, err := gemini.ToGeminiResponsesRequestWithImageURLSchemes(nil, &schemas.BifrostResponsesRequest{ + Provider: schemas.Vertex, + Model: "gemini-3-flash-preview", + Input: []schemas.ResponsesMessage{ + { + Role: schemas.Ptr(schemas.ResponsesInputMessageRoleUser), + Content: &schemas.ResponsesMessageContent{ + ContentBlocks: []schemas.ResponsesMessageContentBlock{ + { + Type: schemas.ResponsesInputMessageContentBlockTypeText, + Text: schemas.Ptr("Describe this image."), + }, + { + Type: schemas.ResponsesInputMessageContentBlockTypeImage, + ResponsesInputMessageContentBlockImage: &schemas.ResponsesInputMessageContentBlockImage{ + ImageURL: schemas.Ptr("gs://my-bucket/xxx.png"), + }, + }, + }, + }, + }, + }, + }, "http", "https", "gs") + + require.NoError(t, err) + require.Len(t, result.Contents, 1) + require.Len(t, result.Contents[0].Parts, 2) + require.NotNil(t, result.Contents[0].Parts[1].FileData) + assert.Equal(t, "gs://my-bucket/xxx.png", result.Contents[0].Parts[1].FileData.FileURI) + assert.Equal(t, "image/png", result.Contents[0].Parts[1].FileData.MIMEType) +} + +// TestVertexGeminiRejectsUnsupportedScheme guards the upper bound of the Vertex +// scheme allowlist: even though Vertex extends Gemini's defaults with "gs", schemes +// outside that set (file://, ftp://, ...) must still be rejected. +func TestVertexGeminiRejectsUnsupportedScheme(t *testing.T) { + _, err := gemini.ToGeminiChatCompletionRequestWithImageURLSchemes(nil, &schemas.BifrostChatRequest{ + Model: "gemini-3-flash-preview", + Input: []schemas.ChatMessage{ + { + Role: schemas.ChatMessageRoleUser, + Content: &schemas.ChatMessageContent{ + ContentBlocks: []schemas.ChatContentBlock{ + { + Type: schemas.ChatContentBlockTypeImage, + ImageURLStruct: &schemas.ChatInputImage{ + URL: "file:///etc/passwd", + }, + }, + }, + }, + }, + }, + }, "http", "https", "gs") + + require.Error(t, err) + assert.Contains(t, err.Error(), `URL scheme "file" is not allowed`) +} diff --git a/core/providers/vllm/vllm.go b/core/providers/vllm/vllm.go index 81d79966910..d24d20b38a5 100644 --- a/core/providers/vllm/vllm.go +++ b/core/providers/vllm/vllm.go @@ -129,7 +129,7 @@ func (provider *VLLMProvider) TextCompletion(ctx *schemas.BifrostContext, key sc provider.client, baseURL+providerUtils.GetPathFromContext(ctx, "/v1/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -147,16 +147,12 @@ func (provider *VLLMProvider) TextCompletionStream(ctx *schemas.BifrostContext, if bifrostErr != nil { return nil, bifrostErr } - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } return openai.HandleOpenAITextCompletionStreaming( ctx, provider.streamingClient, baseURL+providerUtils.GetPathFromContext(ctx, "/v1/completions"), request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -183,13 +179,14 @@ func (provider *VLLMProvider) ChatCompletion(ctx *schemas.BifrostContext, key sc provider.client, baseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), HandleVLLMResponse, nil, + nil, provider.logger, ) } @@ -201,16 +198,12 @@ func (provider *VLLMProvider) ChatCompletionStream(ctx *schemas.BifrostContext, if bifrostErr != nil { return nil, bifrostErr } - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } return openai.HandleOpenAIChatCompletionStreaming( ctx, provider.streamingClient, baseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -222,6 +215,7 @@ func (provider *VLLMProvider) ChatCompletionStream(ctx *schemas.BifrostContext, nil, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -238,7 +232,7 @@ func (provider *VLLMProvider) Embedding(ctx *schemas.BifrostContext, key schemas provider.client, baseURL+providerUtils.GetPathFromContext(ctx, "/v1/embeddings"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -322,7 +316,7 @@ func (provider *VLLMProvider) callVLLMRerankEndpoint( statusCode := resp.StatusCode() if statusCode != fasthttp.StatusOK { rawErrBody := append([]byte(nil), resp.Body()...) - return nil, nil, nil, rawErrBody, statusCode, latency, openai.ParseOpenAIError(resp) + return nil, nil, nil, rawErrBody, statusCode, latency, providerUtils.SetErrorLatency(openai.ParseOpenAIError(resp), latency) } body, err := providerUtils.CheckAndDecodeBody(resp) @@ -373,7 +367,7 @@ func (provider *VLLMProvider) Rerank(ctx *schemas.BifrostContext, key schemas.Ke latency += fallbackLatency } if bifrostErr != nil { - return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, sendBackRawRequest, sendBackRawResponse) + return nil, providerUtils.EnrichError(ctx, bifrostErr, jsonData, responseBody, sendBackRawRequest, sendBackRawResponse, latency) } returnDocuments := request.Params != nil && request.Params.ReturnDocuments != nil && *request.Params.ReturnDocuments @@ -386,6 +380,7 @@ func (provider *VLLMProvider) Rerank(ctx *schemas.BifrostContext, key schemas.Ke responseBody, sendBackRawRequest, sendBackRawResponse, + latency, ) } @@ -489,22 +484,23 @@ func (provider *VLLMProvider) TranscriptionStream(ctx *schemas.BifrostContext, p startTime := time.Now() // Make the request err := provider.streamingClient.Do(req, resp) + latency := time.Since(startTime) if err != nil { defer providerUtils.ReleaseStreamingResponse(ctx, resp) if errors.Is(err, context.Canceled) { - return nil, &schemas.BifrostError{ + return nil, providerUtils.SetErrorLatency(&schemas.BifrostError{ IsBifrostError: false, Error: &schemas.ErrorField{ Type: schemas.Ptr(schemas.RequestCancelled), Message: schemas.ErrRequestCancelled, Error: err, }, - } + }, latency) } if errors.Is(err, fasthttp.ErrTimeout) || errors.Is(err, context.DeadlineExceeded) { - return nil, providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostTimeoutError(schemas.ErrProviderRequestTimedOut, err), latency) } - return nil, providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, err) + return nil, providerUtils.SetErrorLatency(providerUtils.NewBifrostOperationError(schemas.ErrProviderDoRequest, err), latency) } // Store provider response headers in context before status check so error responses also forward them @@ -513,9 +509,11 @@ func (provider *VLLMProvider) TranscriptionStream(ctx *schemas.BifrostContext, p // Check for HTTP errors if resp.StatusCode() != fasthttp.StatusOK { defer providerUtils.ReleaseStreamingResponse(ctx, resp) - return nil, openai.ParseOpenAIError(resp) + return nil, providerUtils.SetErrorLatency(openai.ParseOpenAIError(resp), latency) } + providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) + // Large payload streaming passthrough — pipe raw upstream SSE to client if providerUtils.SetupStreamingPassthrough(ctx, resp) { responseChan := make(chan *schemas.BifrostStreamChunk) @@ -526,8 +524,6 @@ func (provider *VLLMProvider) TranscriptionStream(ctx *schemas.BifrostContext, p // Create response channel responseChan := make(chan *schemas.BifrostStreamChunk, schemas.DefaultStreamBufferSize) - providerUtils.SetStreamIdleTimeoutIfEmpty(ctx, provider.networkConfig.StreamIdleTimeoutInSeconds) - // Start streaming in a goroutine go func() { defer providerUtils.EnsureStreamFinalizerCalled(ctx, postHookSpanFinalizer) @@ -592,7 +588,7 @@ func (provider *VLLMProvider) TranscriptionStream(ctx *schemas.BifrostContext, p _, _, bifrostErr = HandleVLLMResponse(dataBytes, &response, nil, false, false) if bifrostErr != nil { ctx.SetValue(schemas.BifrostContextKeyStreamEndIndicator, true) - providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, body.Bytes(), dataBytes, false, providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse)), responseChan, logger, postHookSpanFinalizer) + providerUtils.ProcessAndSendBifrostError(ctx, postHookRunner, providerUtils.EnrichError(ctx, bifrostErr, body.Bytes(), dataBytes, false, providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), latency), responseChan, logger, postHookSpanFinalizer) return } diff --git a/core/providers/xai/xai.go b/core/providers/xai/xai.go index 0efe56f8ee1..fba2a8f88ec 100644 --- a/core/providers/xai/xai.go +++ b/core/providers/xai/xai.go @@ -91,7 +91,7 @@ func (provider *XAIProvider) TextCompletion(ctx *schemas.BifrostContext, key sch provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.GetProviderKey(), providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -133,13 +133,14 @@ func (provider *XAIProvider) ChatCompletion(ctx *schemas.BifrostContext, key sch provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/chat/completions"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), nil, ParseXAIError, + nil, provider.logger, ) } @@ -149,17 +150,12 @@ func (provider *XAIProvider) ChatCompletion(ctx *schemas.BifrostContext, key sch // Uses xAI's OpenAI-compatible streaming format. // Returns a channel containing BifrostStreamChunk objects representing the stream or an error if the request fails. func (provider *XAIProvider) ChatCompletionStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostChatRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } - // Use shared OpenAI-compatible streaming logic return openai.HandleOpenAIChatCompletionStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+"/v1/chat/completions", request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -171,6 +167,7 @@ func (provider *XAIProvider) ChatCompletionStream(ctx *schemas.BifrostContext, p ParseXAIError, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -183,29 +180,26 @@ func (provider *XAIProvider) Responses(ctx *schemas.BifrostContext, key schemas. provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/responses"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), provider.GetProviderKey(), nil, ParseXAIError, + nil, provider.logger, ) } // ResponsesStream performs a streaming responses request to the xAI API. func (provider *XAIProvider) ResponsesStream(ctx *schemas.BifrostContext, postHookRunner schemas.PostHookRunner, postHookSpanFinalizer func(context.Context), key schemas.Key, request *schemas.BifrostResponsesRequest) (chan *schemas.BifrostStreamChunk, *schemas.BifrostError) { - var authHeader map[string]string - if key.Value.GetValue() != "" { - authHeader = map[string]string{"Authorization": "Bearer " + key.Value.GetValue()} - } return openai.HandleOpenAIResponsesStreaming( ctx, provider.streamingClient, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/responses"), request, - authHeader, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, provider.networkConfig.StreamIdleTimeoutInSeconds, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), @@ -216,6 +210,7 @@ func (provider *XAIProvider) ResponsesStream(ctx *schemas.BifrostContext, postHo ParseXAIError, nil, nil, + nil, provider.logger, postHookSpanFinalizer, ) @@ -388,7 +383,7 @@ func (provider *XAIProvider) Compaction(ctx *schemas.BifrostContext, key schemas provider.client, provider.networkConfig.BaseURL+providerUtils.GetPathFromContext(ctx, "/v1/responses/compact"), request, - key, + openai.BearerAuthHeader(key), provider.networkConfig.ExtraHeaders, providerUtils.ShouldSendBackRawRequest(ctx, provider.sendBackRawRequest), providerUtils.ShouldSendBackRawResponse(ctx, provider.sendBackRawResponse), diff --git a/core/schemas/account.go b/core/schemas/account.go index 9c3f466cac8..98383c99a5d 100644 --- a/core/schemas/account.go +++ b/core/schemas/account.go @@ -124,25 +124,26 @@ func (bl BlackList) Validate() error { // Key represents an API key and its associated configuration for a provider. // It contains the key value, supported models, and a weight for load balancing. type Key struct { - ID string `json:"id"` // The unique identifier for the key (used by bifrost to identify the key) - Name string `json:"name"` // The name of the key (used by users to identify the key, not used by bifrost) - Value SecretVar `json:"value"` // The actual API key value - Models WhiteList `json:"models"` // List of models this key can access - BlacklistedModels BlackList `json:"blacklisted_models"` // List of models this key cannot access - Weight float64 `json:"weight"` // Weight for load balancing between multiple keys - Aliases KeyAliases `json:"aliases,omitempty"` // Mapping of model identifiers to inference profiles - AzureKeyConfig *AzureKeyConfig `json:"azure_key_config,omitempty"` // Azure-specific key configuration - VertexKeyConfig *VertexKeyConfig `json:"vertex_key_config,omitempty"` // Vertex-specific key configuration - BedrockKeyConfig *BedrockKeyConfig `json:"bedrock_key_config,omitempty"` // AWS Bedrock-specific key configuration - VLLMKeyConfig *VLLMKeyConfig `json:"vllm_key_config,omitempty"` // vLLM-specific key configuration - ReplicateKeyConfig *ReplicateKeyConfig `json:"replicate_key_config,omitempty"` // Replicate-specific key configuration - OllamaKeyConfig *OllamaKeyConfig `json:"ollama_key_config,omitempty"` // Ollama-specific key configuration - SGLKeyConfig *SGLKeyConfig `json:"sgl_key_config,omitempty"` // SGLang-specific key configuration - Enabled *bool `json:"enabled,omitempty"` // Whether the key is active (default:true) - UseForBatchAPI *bool `json:"use_for_batch_api,omitempty"` // Whether this key can be used for batch API operations (default:false for new keys, migrated keys default to true) - ConfigHash string `json:"config_hash,omitempty"` // Hash of config.json version, used for change detection - Status KeyStatusType `json:"status,omitempty"` // Status of key - Description string `json:"description,omitempty"` // Description of key + ID string `json:"id"` // The unique identifier for the key (used by bifrost to identify the key) + Name string `json:"name"` // The name of the key (used by users to identify the key, not used by bifrost) + Value SecretVar `json:"value"` // The actual API key value + Models WhiteList `json:"models"` // List of models this key can access + BlacklistedModels BlackList `json:"blacklisted_models"` // List of models this key cannot access + Weight float64 `json:"weight"` // Weight for load balancing between multiple keys + Aliases KeyAliases `json:"aliases,omitempty"` // Mapping of model identifiers to inference profiles + AzureKeyConfig *AzureKeyConfig `json:"azure_key_config,omitempty"` // Azure-specific key configuration + VertexKeyConfig *VertexKeyConfig `json:"vertex_key_config,omitempty"` // Vertex-specific key configuration + BedrockKeyConfig *BedrockKeyConfig `json:"bedrock_key_config,omitempty"` // AWS Bedrock-specific key configuration + BedrockMantleKeyConfig *BedrockMantleKeyConfig `json:"bedrock_mantle_key_config,omitempty"` // Bedrock Mantle-specific key configuration + VLLMKeyConfig *VLLMKeyConfig `json:"vllm_key_config,omitempty"` // vLLM-specific key configuration + ReplicateKeyConfig *ReplicateKeyConfig `json:"replicate_key_config,omitempty"` // Replicate-specific key configuration + OllamaKeyConfig *OllamaKeyConfig `json:"ollama_key_config,omitempty"` // Ollama-specific key configuration + SGLKeyConfig *SGLKeyConfig `json:"sgl_key_config,omitempty"` // SGLang-specific key configuration + Enabled *bool `json:"enabled,omitempty"` // Whether the key is active (default:true) + UseForBatchAPI *bool `json:"use_for_batch_api,omitempty"` // Whether this key can be used for batch API operations (default:false for new keys, migrated keys default to true) + ConfigHash string `json:"config_hash,omitempty"` // Hash of config.json version, used for change detection + Status KeyStatusType `json:"status,omitempty"` // Status of key + Description string `json:"description,omitempty"` // Description of key } // ModelFamily is a typed enum identifying the underlying model family of an alias target. @@ -182,8 +183,8 @@ func (mf *ModelFamily) IsValid() bool { // AzureAliasCfg holds Azure-specific overrides that apply to a single alias. // Each field, when non-nil, overrides the corresponding key-level default. type AzureAliasCfg struct { - APIVersion *string `json:"api_version,omitempty"` // overrides the Azure OpenAI api-version query param for this alias - AnthropicVersion *string `json:"anthropic_version,omitempty"` // overrides the anthropic-version header for Claude-on-Azure deployments + APIVersion *string `json:"api_version,omitempty"` // overrides the Azure OpenAI api-version query param for this alias + AnthropicVersion *string `json:"anthropic_version,omitempty"` // overrides the anthropic-version header for Claude-on-Azure deployments Endpoint *SecretVar `json:"endpoint,omitempty"` // overrides AzureKeyConfig.Endpoint for this alias — lets one credential span deployments on multiple Azure resources } @@ -213,7 +214,7 @@ type AliasConfig struct { ModelName *string `json:"model_name,omitempty"` // canonical model name used for pricing, logging, and 2nd-tier family routing ModelFamily *ModelFamily `json:"model_family,omitempty"` // 1st-tier family routing enum Description string `json:"description,omitempty"` // description of the alias for users to understand its purpose (not used by bifrost) - Region *SecretVar `json:"region,omitempty"` + Region *SecretVar `json:"region,omitempty"` *AzureAliasCfg *VertexAliasCfg @@ -381,6 +382,8 @@ func ResolveFamily(ctx *BifrostContext, fallbackModel string) ModelFamily { switch { case IsAnthropicModel(s): return ModelFamilyAnthropic + case IsOpenAIModel(s): + return ModelFamilyOpenAI case IsMistralModel(s): return ModelFamilyMistral // Imagen and Veo are checked before Gemini as a defensive ordering: @@ -442,6 +445,10 @@ func IsAnthropicModelFamily(ctx *BifrostContext, model string) bool { return ResolveFamily(ctx, model) == ModelFamilyAnthropic } +func IsOpenAIModelFamily(ctx *BifrostContext, model string) bool { + return ResolveFamily(ctx, model) == ModelFamilyOpenAI +} + // IsMistralModelFamily reports whether the current attempt resolves to the // Mistral model family. See IsAnthropicModelFamily for usage notes. func IsMistralModelFamily(ctx *BifrostContext, model string) bool { @@ -608,10 +615,10 @@ const ( type AzureKeyConfig struct { Endpoint SecretVar `json:"endpoint"` // Azure service endpoint URL - ClientID *SecretVar `json:"client_id,omitempty"` // Azure client ID for authentication - ClientSecret *SecretVar `json:"client_secret,omitempty"` // Azure client secret for authentication - TenantID *SecretVar `json:"tenant_id,omitempty"` // Azure tenant ID for authentication - Scopes []string `json:"scopes,omitempty"` + ClientID *SecretVar `json:"client_id,omitempty"` // Azure client ID for authentication + ClientSecret *SecretVar `json:"client_secret,omitempty"` // Azure client secret for authentication + TenantID *SecretVar `json:"tenant_id,omitempty"` // Azure tenant ID for authentication + Scopes []string `json:"scopes,omitempty"` } // VertexKeyConfig represents the Vertex-specific configuration. @@ -657,12 +664,31 @@ type BedrockKeyConfig struct { // NOTE: To use Bedrock IAM role authentication, set both AccessKey and SecretKey to empty strings. // To use Bedrock API Key authentication, set Value in Key struct instead. +// BedrockMantleKeyConfig represents the Bedrock Mantle-specific configuration. Mantle serves +// Claude (native-Anthropic Messages), OpenAI-compatible, and Gemma models on the +// bedrock-mantle.{region}.api.aws host. It carries only the credentials and region the mantle +// endpoints use; it intentionally omits the inference-profile ARN and batch S3 config, which +// apply only to the Converse/bedrock-runtime surface. +type BedrockMantleKeyConfig struct { + AccessKey SecretVar `json:"access_key,omitempty"` // AWS access key for SigV4 authentication + SecretKey SecretVar `json:"secret_key,omitempty"` // AWS secret access key for SigV4 authentication + SessionToken *SecretVar `json:"session_token,omitempty"` // AWS session token for temporary credentials + Region *SecretVar `json:"region,omitempty"` // AWS region used to build the bedrock-mantle endpoint host + // IAM role for STS AssumeRole + RoleARN *SecretVar `json:"role_arn,omitempty"` + ExternalID *SecretVar `json:"external_id,omitempty"` + RoleSessionName *SecretVar `json:"session_name,omitempty"` +} + +// NOTE: To use Bedrock Mantle IAM role authentication, set both AccessKey and SecretKey to empty +// strings. To use Bedrock Mantle API Key authentication, set Value in the Key struct instead. + // VLLMKeyConfig represents the vLLM-specific key configuration. // It allows each key to target a different vLLM server URL and model name, // enabling per-key routing and round-robin load balancing across multiple vLLM instances. type VLLMKeyConfig struct { URL SecretVar `json:"url"` // VLLM server base URL (required, supports env. prefix) - ModelName string `json:"model_name"` // Exact model name served on this VLLM instance (used for key selection) + ModelName string `json:"model_name"` // Exact model name served on this VLLM instance (used for key selection) } // ReplicateKeyConfig represents the Replicate-specific key configuration. diff --git a/core/schemas/account_test.go b/core/schemas/account_test.go index 21103896470..35e916dbd4a 100644 --- a/core/schemas/account_test.go +++ b/core/schemas/account_test.go @@ -457,6 +457,42 @@ func TestResolveFamilyPrecedence(t *testing.T) { } } +func TestIsOpenAIModel(t *testing.T) { + cases := []struct { + model string + want bool + }{ + // gpt / embeddings (pre-existing behavior) + {"gpt-4o", true}, + {"gpt-5", true}, + {"text-embedding-3-large", true}, + // o-series reasoning families + {"o1", true}, + {"o1-preview", true}, + {"o1-mini", true}, + {"o3", true}, + {"o3-mini", true}, + {"o4-mini", true}, + {"openai/o3", true}, + {"openai:o4-mini", true}, + // non-OpenAI / false-positive guards + {"claude-3-5-sonnet", false}, + {"mistral-large-2407", false}, + {"co1", false}, // must not match "o1" mid-word + {"o3x", false}, // version digit must be terminal or followed by "-" + {"model-o3", false}, // "o3" not at the start of the (post-prefix) name + {"", false}, + {"o", false}, + } + for _, c := range cases { + t.Run(c.model, func(t *testing.T) { + if got := IsOpenAIModel(c.model); got != c.want { + t.Fatalf("IsOpenAIModel(%q): got %v, want %v", c.model, got, c.want) + } + }) + } +} + func TestResolveCanonicalModelPrecedence(t *testing.T) { // Helper to build a BifrostContext carrying a ResolvedAlias. withAlias := func(ra *ResolvedAlias) *BifrostContext { diff --git a/core/schemas/async.go b/core/schemas/async.go index 6166b7f98f9..0d368cae2d0 100644 --- a/core/schemas/async.go +++ b/core/schemas/async.go @@ -1,10 +1,19 @@ package schemas -import "time" +import ( + "database/sql/driver" + "time" +) // AsyncJobStatus represents the status of an async job type AsyncJobStatus string +// Value implements driver.Valuer so database drivers that append typed +// column values (e.g. clickhouse-go batch inserts) can serialize the type. +func (s AsyncJobStatus) Value() (driver.Value, error) { + return string(s), nil +} + const ( AsyncJobStatusPending AsyncJobStatus = "pending" AsyncJobStatusProcessing AsyncJobStatus = "processing" diff --git a/core/schemas/bifrost.go b/core/schemas/bifrost.go index b1448456bfb..b4f8dc683cc 100644 --- a/core/schemas/bifrost.go +++ b/core/schemas/bifrost.go @@ -2,6 +2,7 @@ package schemas import ( + "database/sql/driver" "encoding/json" "errors" "fmt" @@ -41,32 +42,34 @@ type BifrostConfig struct { type ModelProvider string const ( - OpenAI ModelProvider = "openai" - Azure ModelProvider = "azure" - Anthropic ModelProvider = "anthropic" - Bedrock ModelProvider = "bedrock" - Cohere ModelProvider = "cohere" - Vertex ModelProvider = "vertex" - Mistral ModelProvider = "mistral" - Ollama ModelProvider = "ollama" - OpencodeGo ModelProvider = "opencode-go" - OpencodeZen ModelProvider = "opencode-zen" - Groq ModelProvider = "groq" - SGL ModelProvider = "sgl" - Parasail ModelProvider = "parasail" - Perplexity ModelProvider = "perplexity" - Cerebras ModelProvider = "cerebras" - Gemini ModelProvider = "gemini" - OpenRouter ModelProvider = "openrouter" - Elevenlabs ModelProvider = "elevenlabs" - HuggingFace ModelProvider = "huggingface" - Nebius ModelProvider = "nebius" - XAI ModelProvider = "xai" - Replicate ModelProvider = "replicate" - VLLM ModelProvider = "vllm" - Runway ModelProvider = "runway" - Runware ModelProvider = "runware" - Fireworks ModelProvider = "fireworks" + OpenAI ModelProvider = "openai" + Azure ModelProvider = "azure" + Anthropic ModelProvider = "anthropic" + Bedrock ModelProvider = "bedrock" + BedrockMantle ModelProvider = "bedrock_mantle" + Cohere ModelProvider = "cohere" + Vertex ModelProvider = "vertex" + Mistral ModelProvider = "mistral" + Ollama ModelProvider = "ollama" + OpencodeGo ModelProvider = "opencode-go" + OpencodeZen ModelProvider = "opencode-zen" + Groq ModelProvider = "groq" + SGL ModelProvider = "sgl" + Parasail ModelProvider = "parasail" + Perplexity ModelProvider = "perplexity" + Cerebras ModelProvider = "cerebras" + DeepSeek ModelProvider = "deepseek" + Gemini ModelProvider = "gemini" + OpenRouter ModelProvider = "openrouter" + Elevenlabs ModelProvider = "elevenlabs" + HuggingFace ModelProvider = "huggingface" + Nebius ModelProvider = "nebius" + XAI ModelProvider = "xai" + Replicate ModelProvider = "replicate" + VLLM ModelProvider = "vllm" + Runway ModelProvider = "runway" + Runware ModelProvider = "runware" + Fireworks ModelProvider = "fireworks" ) // SupportedBaseProviders is the list of base providers allowed for custom providers. @@ -85,8 +88,10 @@ var StandardProviders = []ModelProvider{ Anthropic, Azure, Bedrock, + BedrockMantle, Cerebras, Cohere, + DeepSeek, Gemini, Groq, Mistral, @@ -113,6 +118,12 @@ var StandardProviders = []ModelProvider{ // RequestType represents the type of request being made to a provider. type RequestType string +// Value implements driver.Valuer so database drivers that append typed +// column values (e.g. clickhouse-go batch inserts) can serialize the type. +func (r RequestType) Value() (driver.Value, error) { + return string(r), nil +} + const ( ListModelsRequest RequestType = "list_models" TextCompletionRequest RequestType = "text_completion" @@ -121,6 +132,10 @@ const ( ChatCompletionStreamRequest RequestType = "chat_completion_stream" ResponsesRequest RequestType = "responses" ResponsesStreamRequest RequestType = "responses_stream" + ResponsesRetrieveRequest RequestType = "responses_retrieve" + ResponsesDeleteRequest RequestType = "responses_delete" + ResponsesCancelRequest RequestType = "responses_cancel" + ResponsesInputItemsRequest RequestType = "responses_input_items" EmbeddingRequest RequestType = "embedding" SpeechRequest RequestType = "speech" SpeechStreamRequest RequestType = "speech_stream" @@ -243,7 +258,7 @@ const ( BifrostContextKeyFallbackIndex BifrostContextKey = "bifrost-fallback-index" // int (to store the fallback index (set by bifrost - DO NOT SET THIS MANUALLY)) 0 for primary, 1 for first fallback, etc. BifrostContextKeyResolvedAlias BifrostContextKey = "bifrost-resolved-alias" // *ResolvedAlias (set by bifrost after key-level alias resolution — providers read this for model_family routing and provider-specific overrides; nil/absent when no alias matched) BifrostContextKeyStreamEndIndicator BifrostContextKey = "bifrost-stream-end-indicator" // bool (set by bifrost - DO NOT SET THIS MANUALLY)) - BifrostContextKeyStreamGated BifrostContextKey = "bifrost-stream-gated" // bool (set by ctx.PauseStream/ResumeStream/EndStream when a plugin first engages the pause/resume gate; provider helpers use this as a fast-path check to skip Tracer.GateSend on streams that never engage the gate) + BifrostContextKeyStreamGated BifrostContextKey = "bifrost-stream-gated" // bool (set by ctx.PauseStream/ResumeStream/EndStream when a plugin first engages the pause/resume gate; provider helpers use this as a fast-path check to skip Tracer.GateSend on streams that never engage the gate) BifrostContextKeyStreamIdleTimeout BifrostContextKey = "bifrost-stream-idle-timeout" // time.Duration (per-chunk idle timeout for streaming) BifrostContextKeySkipKeySelection BifrostContextKey = "bifrost-skip-key-selection" // bool (will pass an empty key to the provider) BifrostContextKeyExtraHeaders BifrostContextKey = "bifrost-extra-headers" // map[string][]string @@ -369,11 +384,11 @@ const ( // RoutingEngine constants const ( - RoutingEngineGovernance = "governance" - RoutingEngineRoutingRule = "routing-rule" - RoutingEngineLoadbalancing = "loadbalancing" - RoutingEngineModelCatalog = "model-catalog" - RoutingEngineCircuitBreaker = "circuit-breaker" + RoutingEngineGovernance = "governance" + RoutingEngineRoutingRule = "routing-rule" + RoutingEngineLoadbalancing = "loadbalancing" + RoutingEngineModelCatalog = "model-catalog" + RoutingEngineCircuitBreaker = "circuit-breaker" // RoutingEngineCore represents the Bifrost core orchestrator's own // routing decisions — primarily fallback transitions. Emitted when the // primary attempt fails and core advances through the fallback chain so @@ -478,6 +493,10 @@ type BifrostRequest struct { TextCompletionRequest *BifrostTextCompletionRequest ChatRequest *BifrostChatRequest ResponsesRequest *BifrostResponsesRequest + ResponsesRetrieveRequest *BifrostResponsesRetrieveRequest + ResponsesDeleteRequest *BifrostResponsesDeleteRequest + ResponsesCancelRequest *BifrostResponsesCancelRequest + ResponsesInputItemsRequest *BifrostResponsesInputItemsRequest CountTokensRequest *BifrostResponsesRequest CompactionRequest *BifrostCompactionRequest EmbeddingRequest *BifrostEmbeddingRequest @@ -533,6 +552,14 @@ func (br *BifrostRequest) GetRequestFields() (provider ModelProvider, model stri return br.ChatRequest.Provider, br.ChatRequest.Model, br.ChatRequest.Fallbacks case br.ResponsesRequest != nil: return br.ResponsesRequest.Provider, br.ResponsesRequest.Model, br.ResponsesRequest.Fallbacks + case br.ResponsesRetrieveRequest != nil: + return br.ResponsesRetrieveRequest.Provider, "", nil + case br.ResponsesDeleteRequest != nil: + return br.ResponsesDeleteRequest.Provider, "", nil + case br.ResponsesCancelRequest != nil: + return br.ResponsesCancelRequest.Provider, "", nil + case br.ResponsesInputItemsRequest != nil: + return br.ResponsesInputItemsRequest.Provider, "", nil case br.CountTokensRequest != nil: return br.CountTokensRequest.Provider, br.CountTokensRequest.Model, br.CountTokensRequest.Fallbacks case br.CompactionRequest != nil: @@ -676,6 +703,14 @@ func (br *BifrostRequest) SetProvider(provider ModelProvider) { br.ChatRequest.Provider = provider case br.ResponsesRequest != nil: br.ResponsesRequest.Provider = provider + case br.ResponsesRetrieveRequest != nil: + br.ResponsesRetrieveRequest.Provider = provider + case br.ResponsesDeleteRequest != nil: + br.ResponsesDeleteRequest.Provider = provider + case br.ResponsesCancelRequest != nil: + br.ResponsesCancelRequest.Provider = provider + case br.ResponsesInputItemsRequest != nil: + br.ResponsesInputItemsRequest.Provider = provider case br.CountTokensRequest != nil: br.CountTokensRequest.Provider = provider case br.CompactionRequest != nil: @@ -817,6 +852,14 @@ func (br *BifrostRequest) SetRawRequestBody(rawRequestBody []byte) { br.ChatRequest.RawRequestBody = rawRequestBody case br.ResponsesRequest != nil: br.ResponsesRequest.RawRequestBody = rawRequestBody + case br.ResponsesRetrieveRequest != nil: + br.ResponsesRetrieveRequest.RawRequestBody = rawRequestBody + case br.ResponsesDeleteRequest != nil: + br.ResponsesDeleteRequest.RawRequestBody = rawRequestBody + case br.ResponsesCancelRequest != nil: + br.ResponsesCancelRequest.RawRequestBody = rawRequestBody + case br.ResponsesInputItemsRequest != nil: + br.ResponsesInputItemsRequest.RawRequestBody = rawRequestBody case br.CountTokensRequest != nil: br.CountTokensRequest.RawRequestBody = rawRequestBody case br.CompactionRequest != nil: @@ -977,6 +1020,8 @@ type BifrostResponse struct { ChatResponse *BifrostChatResponse ResponsesResponse *BifrostResponsesResponse ResponsesStreamResponse *BifrostResponsesStreamResponse + ResponsesDeleteResponse *BifrostResponsesDeleteResponse + ResponsesInputItemsResponse *BifrostResponsesInputItemsResponse CountTokensResponse *BifrostCountTokensResponse CompactionResponse *BifrostCompactionResponse EmbeddingResponse *BifrostEmbeddingResponse @@ -1032,6 +1077,10 @@ func (r *BifrostResponse) GetExtraFields() *BifrostResponseExtraFields { return &r.ResponsesResponse.ExtraFields case r.ResponsesStreamResponse != nil: return &r.ResponsesStreamResponse.ExtraFields + case r.ResponsesDeleteResponse != nil: + return &r.ResponsesDeleteResponse.ExtraFields + case r.ResponsesInputItemsResponse != nil: + return &r.ResponsesInputItemsResponse.ExtraFields case r.CountTokensResponse != nil: return &r.CountTokensResponse.ExtraFields case r.CompactionResponse != nil: @@ -1252,6 +1301,16 @@ func (r *BifrostResponse) PopulateExtraFields(requestType RequestType, provider r.ResponsesResponse.ExtraFields.Provider = provider r.ResponsesResponse.ExtraFields.OriginalModelRequested = originalModelRequested r.ResponsesResponse.ExtraFields.ResolvedModelUsed = resolvedModel + case r.ResponsesDeleteResponse != nil: + r.ResponsesDeleteResponse.ExtraFields.RequestType = requestType + r.ResponsesDeleteResponse.ExtraFields.Provider = provider + r.ResponsesDeleteResponse.ExtraFields.OriginalModelRequested = originalModelRequested + r.ResponsesDeleteResponse.ExtraFields.ResolvedModelUsed = resolvedModel + case r.ResponsesInputItemsResponse != nil: + r.ResponsesInputItemsResponse.ExtraFields.RequestType = requestType + r.ResponsesInputItemsResponse.ExtraFields.Provider = provider + r.ResponsesInputItemsResponse.ExtraFields.OriginalModelRequested = originalModelRequested + r.ResponsesInputItemsResponse.ExtraFields.ResolvedModelUsed = resolvedModel case r.ResponsesStreamResponse != nil: r.ResponsesStreamResponse.ExtraFields.RequestType = requestType r.ResponsesStreamResponse.ExtraFields.Provider = provider @@ -1876,6 +1935,7 @@ type BifrostErrorExtraFields struct { RawResponse interface{} `json:"raw_response,omitempty"` ConvertedRequestType RequestType `json:"converted_request_type,omitempty"` DroppedCompatPluginParams []string `json:"dropped_compat_plugin_params,omitempty"` + Latency int64 `json:"latency,omitempty"` // in milliseconds KeyStatuses []KeyStatus `json:"key_statuses,omitempty"` MCPAuthRequired *MCPAuthRequiredError `json:"mcp_auth_required,omitempty"` // Set when a per-user MCP tool requires the caller to complete an inline auth flow (OAuth or headers) // BilledUsage carries provider-reported token usage that was consumed even diff --git a/core/schemas/chatcompletions.go b/core/schemas/chatcompletions.go index 0296dd04ce9..96f3fd1028c 100644 --- a/core/schemas/chatcompletions.go +++ b/core/schemas/chatcompletions.go @@ -40,7 +40,7 @@ type BifrostChatResponse struct { Model string `json:"model"` Object string `json:"object"` // "chat.completion" or "chat.completion.chunk" ServiceTier *BifrostServiceTier `json:"service_tier,omitempty"` - Speed *string `json:"speed,omitempty"` // "fast" | "standard" — speed actually served (Anthropic fast mode); drives fast-mode billing + Speed *string `json:"speed,omitempty"` // "fast" | "standard" — speed actually served (Anthropic fast mode); drives fast-mode billing Diagnostics *CacheDiagnostics `json:"diagnostics,omitempty"` // Anthropic cache diagnostics (cache-diagnosis-2026-04-07); first prompt-cache prefix divergence point SystemFingerprint string `json:"system_fingerprint"` Usage *BifrostLLMUsage `json:"usage"` @@ -202,7 +202,7 @@ type ChatParameters struct { Prediction *ChatPrediction `json:"prediction,omitempty"` // Predicted output content (OpenAI only) PresencePenalty *float64 `json:"presence_penalty,omitempty"` // Penalizes repeated tokens PromptCacheKey *string `json:"prompt_cache_key,omitempty"` // Prompt cache key - PromptCacheRetention *string `json:"prompt_cache_retention,omitempty"` // Prompt cache retention ("in-memory" or "24h") + PromptCacheRetention *string `json:"prompt_cache_retention,omitempty"` // Prompt cache retention ("in_memory" or "24h") Reasoning *ChatReasoning `json:"reasoning,omitempty"` // Reasoning parameters ResponseFormat *interface{} `json:"response_format,omitempty"` // Format for the response SafetyIdentifier *string `json:"safety_identifier,omitempty"` // Safety identifier @@ -1666,7 +1666,9 @@ func (d *ChatPromptTokensDetails) UnmarshalJSON(data []byte) error { return nil } -// MarshalJSON emits cached_tokens (read+write) alongside the individual fields for OpenAI spec compatibility. +// MarshalJSON emits cached_tokens (reads only, per the OpenAI spec and mirroring UnmarshalJSON above) alongside the individual fields. +// Cache writes are reported separately via cached_write_tokens and are excluded from cached_tokens so that +// OpenAI-spec consumers do not price cache writes as cache reads. func (d ChatPromptTokensDetails) MarshalJSON() ([]byte, error) { type raw struct { TextTokens int `json:"text_tokens,omitempty"` @@ -1684,7 +1686,7 @@ func (d ChatPromptTokensDetails) MarshalJSON() ([]byte, error) { CachedReadTokens: d.CachedReadTokens, CachedWriteTokens: d.CachedWriteTokens, CachedWriteTokenDetails: d.CachedWriteTokenDetails, - CachedTokens: d.CachedReadTokens + d.CachedWriteTokens, + CachedTokens: d.CachedReadTokens, }) } diff --git a/core/schemas/images.go b/core/schemas/images.go index 0d9ed181f8c..34279c0aa3b 100644 --- a/core/schemas/images.go +++ b/core/schemas/images.go @@ -60,7 +60,7 @@ type BifrostImageGenerationResponse struct { *ImageGenerationResponseParameters Usage *ImageUsage `json:"usage,omitempty"` - ExtraFields BifrostResponseExtraFields `json:"extra_fields,omitempty"` + ExtraFields BifrostResponseExtraFields `json:"extra_fields"` } // BackfillParams populates response fields from the original request that are needed @@ -121,22 +121,6 @@ func (r *BifrostImageGenerationResponse) BackfillParams(req *BifrostRequest) { } } -// getModelFromRequest extracts the model from any image-related request. -func getModelFromRequest(req *BifrostRequest) string { - if req == nil { - return "" - } - switch { - case req.ImageGenerationRequest != nil: - return req.ImageGenerationRequest.Model - case req.ImageEditRequest != nil: - return req.ImageEditRequest.Model - case req.ImageVariationRequest != nil: - return req.ImageVariationRequest.Model - } - return "" -} - // getNumInputImagesSizeQualityAndAspectRatioFromRequest extracts request params for cost // calculation and logging. Quality is only returned when it is one of low, medium, high, auto. // AspectRatio is only carried by image generation requests. @@ -218,7 +202,7 @@ type ImageUsage struct { TotalTokens int `json:"total_tokens,omitempty"` OutputTokens int `json:"output_tokens,omitempty"` // Always image tokens unless OutputTokensDetails is not nil OutputTokensDetails *ImageTokenDetails `json:"output_tokens_details,omitempty"` - NumInputImages int `json:"num_input_images,omitempty"` // Number of input images from the request (populated by Bifrost) + NumInputImages int `json:"-"` // Number of input images from the request (populated by Bifrost) } type ImageTokenDetails struct { @@ -267,7 +251,7 @@ type BifrostImageGenerationStreamResponse struct { Error *BifrostError `json:"error,omitempty"` RawRequest string `json:"-"` RawResponse string `json:"-"` - ExtraFields BifrostResponseExtraFields `json:"extra_fields,omitempty"` + ExtraFields BifrostResponseExtraFields `json:"extra_fields"` } // BackfillParams populates response fields from the original request that are needed diff --git a/core/schemas/mcp.go b/core/schemas/mcp.go index bae3531bc3d..c8822e10259 100644 --- a/core/schemas/mcp.go +++ b/core/schemas/mcp.go @@ -10,6 +10,7 @@ import ( "errors" "fmt" "io" + "math" "math/big" "net/http" "strconv" @@ -313,9 +314,10 @@ type MCPClientConfig struct { // - nil/omitted => treated as [] (no tools) // - ["tool1", "tool2"] => auto-execute only the specified tools // Note: If a tool is in ToolsToAutoExecute but not in ToolsToExecute, it will be skipped. - IsPingAvailable *bool `json:"is_ping_available,omitempty"` // Whether the MCP server supports ping for health checks (nil/true = ping; false = listTools). Defaults to true. - ToolSyncInterval time.Duration `json:"tool_sync_interval,omitempty"` // Per-client override for tool sync interval (0 = use global, negative = disabled) - ToolPricing map[string]float64 `json:"tool_pricing,omitempty"` // Tool pricing for each tool (cost per execution) + IsPingAvailable *bool `json:"is_ping_available,omitempty"` // Whether the MCP server supports ping for health checks (nil/true = ping; false = listTools). Defaults to true. + ToolSyncInterval time.Duration `json:"tool_sync_interval,omitempty"` // Per-client override for tool sync interval (0 = use global, negative = disabled) + ToolExecutionTimeout time.Duration `json:"tool_execution_timeout,omitempty"` // Per-client override for tool execution timeout (0 = use global from tool_manager_config) + ToolPricing map[string]float64 `json:"tool_pricing,omitempty"` // Tool pricing for each tool (cost per execution) Disabled bool `json:"disabled"` // Whether the client is intentionally disabled (stops connection and workers) ConfigHash string `json:"-"` // Config hash for reconciliation (not serialized) AllowOnAllVirtualKeys bool `json:"allow_on_all_virtual_keys"` // Whether to allow the MCP client to run on all virtual keys @@ -325,12 +327,14 @@ type MCPClientConfig struct { DiscoveredToolNameMapping map[string]string `json:"-"` // Mapping from sanitized tool names to original MCP names } -// UnmarshalJSON supports Go duration strings (e.g. "10m") for tool_sync_interval. -// Numeric values remain supported for backward compatibility (treated as raw nanoseconds). +// UnmarshalJSON supports Go duration strings (e.g. "10m") for tool_sync_interval and +// tool_execution_timeout. Numeric values are treated as raw nanoseconds for tool_sync_interval +// and as seconds for tool_execution_timeout (matching tool_manager_config behaviour). func (c *MCPClientConfig) UnmarshalJSON(data []byte) error { type alias MCPClientConfig aux := &struct { - ToolSyncInterval *json.Number `json:"tool_sync_interval,omitempty"` + ToolSyncInterval *json.Number `json:"tool_sync_interval,omitempty"` + ToolExecutionTimeout *json.RawMessage `json:"tool_execution_timeout,omitempty"` *alias }{alias: (*alias)(c)} @@ -340,36 +344,99 @@ func (c *MCPClientConfig) UnmarshalJSON(data []byte) error { if err := decoder.Decode(&struct{}{}); !errors.Is(err, io.EOF) { return errors.New("trailing JSON data") } - if aux.ToolSyncInterval == nil { - return nil + if aux.ToolSyncInterval != nil { + dur, parseErr := parseFlexibleDurationField(*aux.ToolSyncInterval, "tool_sync_interval") + if parseErr != nil { + return parseErr + } + c.ToolSyncInterval = dur } - dur, parseErr := parseFlexibleDurationField(*aux.ToolSyncInterval, "tool_sync_interval") - if parseErr != nil { - return parseErr + if aux.ToolExecutionTimeout != nil { + dur, err := parseToolExecutionTimeoutField(*aux.ToolExecutionTimeout) + if err != nil { + return err + } + c.ToolExecutionTimeout = dur } - c.ToolSyncInterval = dur return nil } // Allow Go duration strings while keeping numeric tokens as json.Number. + // ToolExecutionTimeout uses *json.RawMessage (not *string) so that integer + // values like 60 remain valid even when tool_sync_interval is a string. auxStr := &struct { - ToolSyncInterval *string `json:"tool_sync_interval,omitempty"` + ToolSyncInterval *string `json:"tool_sync_interval,omitempty"` + ToolExecutionTimeout *json.RawMessage `json:"tool_execution_timeout,omitempty"` *alias }{alias: (*alias)(c)} if err := json.Unmarshal(data, auxStr); err != nil { return err } - if auxStr.ToolSyncInterval == nil { - return nil + if auxStr.ToolSyncInterval != nil { + dur, err := parseFlexibleDurationField(*auxStr.ToolSyncInterval, "tool_sync_interval") + if err != nil { + return err + } + c.ToolSyncInterval = dur } - dur, err := parseFlexibleDurationField(*auxStr.ToolSyncInterval, "tool_sync_interval") - if err != nil { - return err + if auxStr.ToolExecutionTimeout != nil { + dur, err := parseToolExecutionTimeoutField(*auxStr.ToolExecutionTimeout) + if err != nil { + return err + } + c.ToolExecutionTimeout = dur } - c.ToolSyncInterval = dur return nil } +// parseToolExecutionTimeoutField parses a tool_execution_timeout JSON value. +// Accepts a Go duration string (e.g. "30s") or a bare integer treated as seconds. +// Rejects negative values and integers that would overflow time.Duration. +func parseToolExecutionTimeoutField(raw json.RawMessage) (time.Duration, error) { + if len(raw) > 0 && raw[0] == '"' { + var s string + if err := json.Unmarshal(raw, &s); err != nil { + return 0, fmt.Errorf("invalid tool_execution_timeout: %w", err) + } + dur, err := time.ParseDuration(s) + if err != nil { + return 0, fmt.Errorf("invalid tool_execution_timeout %q: %w", s, err) + } + if dur < 0 { + return 0, fmt.Errorf("invalid tool_execution_timeout: value must be >= 0, got %v", dur) + } + return dur, nil + } + var n int64 + if err := json.Unmarshal(raw, &n); err != nil { + return 0, fmt.Errorf("invalid tool_execution_timeout: expected a duration string (e.g. \"30s\") or integer seconds: %w", err) + } + if n < 0 { + return 0, fmt.Errorf("invalid tool_execution_timeout: value must be >= 0, got %d", n) + } + const maxTimeoutSeconds = math.MaxInt64 / int64(time.Second) + if n > maxTimeoutSeconds { + return 0, fmt.Errorf("invalid tool_execution_timeout: value %d seconds overflows duration (max %d)", n, maxTimeoutSeconds) + } + return time.Duration(n) * time.Second, nil +} + +// MarshalJSON emits tool_execution_timeout as a duration string so it round-trips +// correctly — default time.Duration marshaling emits nanoseconds, but UnmarshalJSON +// treats bare integers as seconds. +func (c MCPClientConfig) MarshalJSON() ([]byte, error) { + type alias MCPClientConfig + type shadow struct { + ToolExecutionTimeout string `json:"tool_execution_timeout,omitempty"` + *alias + } + s := shadow{alias: (*alias)(&c)} + if c.ToolExecutionTimeout > 0 { + s.ToolExecutionTimeout = c.ToolExecutionTimeout.String() + } + return json.Marshal(s) +} + func parseFlexibleDurationField(v any, fieldName string) (time.Duration, error) { switch t := v.(type) { case string: diff --git a/core/schemas/mcp_json_test.go b/core/schemas/mcp_json_test.go index 9ee396a02b7..9808104b974 100644 --- a/core/schemas/mcp_json_test.go +++ b/core/schemas/mcp_json_test.go @@ -61,3 +61,71 @@ func TestMCPConfigUnmarshalToolSyncIntervalRejectsFractionalNumber(t *testing.T) } } +func TestMCPClientConfigUnmarshalToolExecutionTimeoutString(t *testing.T) { + raw := []byte(`{"name":"demo","connection_type":"http","tool_execution_timeout":"45s"}`) + var cfg MCPClientConfig + if err := sonic.Unmarshal(raw, &cfg); err != nil { + t.Fatalf("unexpected unmarshal error: %v", err) + } + if cfg.ToolExecutionTimeout != 45*time.Second { + t.Fatalf("expected 45s, got %v", cfg.ToolExecutionTimeout) + } +} + +func TestMCPClientConfigUnmarshalToolExecutionTimeoutInteger(t *testing.T) { + raw := []byte(`{"name":"demo","connection_type":"http","tool_execution_timeout":60}`) + var cfg MCPClientConfig + if err := sonic.Unmarshal(raw, &cfg); err != nil { + t.Fatalf("unexpected unmarshal error: %v", err) + } + if cfg.ToolExecutionTimeout != 60*time.Second { + t.Fatalf("expected 60s, got %v", cfg.ToolExecutionTimeout) + } +} + +func TestMCPClientConfigUnmarshalToolExecutionTimeoutNotSet(t *testing.T) { + raw := []byte(`{"name":"demo","connection_type":"http"}`) + var cfg MCPClientConfig + if err := sonic.Unmarshal(raw, &cfg); err != nil { + t.Fatalf("unexpected unmarshal error: %v", err) + } + if cfg.ToolExecutionTimeout != 0 { + t.Fatalf("expected 0 (use global), got %v", cfg.ToolExecutionTimeout) + } +} + +func TestMCPClientConfigUnmarshalToolExecutionTimeoutExplicitZero(t *testing.T) { + raw := []byte(`{"name":"demo","connection_type":"http","tool_execution_timeout":0}`) + var cfg MCPClientConfig + if err := sonic.Unmarshal(raw, &cfg); err != nil { + t.Fatalf("unexpected unmarshal error: %v", err) + } + if cfg.ToolExecutionTimeout != 0 { + t.Fatalf("expected 0 (use global), got %v", cfg.ToolExecutionTimeout) + } +} + +func TestMCPClientConfigUnmarshalToolExecutionTimeoutInvalidString(t *testing.T) { + raw := []byte(`{"name":"demo","connection_type":"http","tool_execution_timeout":"not-a-duration"}`) + var cfg MCPClientConfig + if err := sonic.Unmarshal(raw, &cfg); err == nil { + t.Fatal("expected unmarshal error for invalid duration, got nil") + } +} + +func TestMCPClientConfigUnmarshalToolExecutionTimeoutNegativeInteger(t *testing.T) { + raw := []byte(`{"name":"demo","connection_type":"http","tool_execution_timeout":-30}`) + var cfg MCPClientConfig + if err := sonic.Unmarshal(raw, &cfg); err == nil { + t.Fatal("expected unmarshal error for negative timeout, got nil") + } +} + +func TestMCPClientConfigUnmarshalToolExecutionTimeoutNegativeString(t *testing.T) { + raw := []byte(`{"name":"demo","connection_type":"http","tool_execution_timeout":"-30s"}`) + var cfg MCPClientConfig + if err := sonic.Unmarshal(raw, &cfg); err == nil { + t.Fatal("expected unmarshal error for negative duration string, got nil") + } +} + diff --git a/core/schemas/models.go b/core/schemas/models.go index b38f79aab01..f36cc2deb1f 100644 --- a/core/schemas/models.go +++ b/core/schemas/models.go @@ -151,12 +151,13 @@ type Model struct { CanonicalSlug *string `json:"canonical_slug,omitempty"` Name *string `json:"name,omitempty"` NormalizedName *string `json:"normalized_name,omitempty"` // Human-readable name derived from the datasheet base_model (e.g. "Claude Sonnet 4.5") - Alias *string `json:"alias,omitempty"` // Provider API identifier this model alias maps to (e.g. Azure deployment name, Bedrock ARN) + Alias *string `json:"alias,omitempty"` // Provider API identifier this model alias maps to (e.g. Azure deployment name, Bedrock ARN) Created *int64 `json:"created,omitempty"` ContextLength *int `json:"context_length,omitempty"` MaxInputTokens *int `json:"max_input_tokens,omitempty"` MaxOutputTokens *int `json:"max_output_tokens,omitempty"` Architecture *Architecture `json:"architecture,omitempty"` + IsDeprecated bool `json:"is_deprecated,omitempty"` Pricing *Pricing `json:"pricing,omitempty"` TopProvider *TopProvider `json:"top_provider,omitempty"` PerRequestLimits *PerRequestLimits `json:"per_request_limits,omitempty"` diff --git a/core/schemas/provider.go b/core/schemas/provider.go index b1da72e2519..c65facded0e 100644 --- a/core/schemas/provider.go +++ b/core/schemas/provider.go @@ -58,7 +58,7 @@ type NetworkConfig struct { RetryBackoffInitial time.Duration `json:"retry_backoff_initial"` // Initial backoff duration (stored as nanoseconds, JSON as milliseconds) RetryBackoffMax time.Duration `json:"retry_backoff_max"` // Maximum backoff duration (stored as nanoseconds, JSON as milliseconds) InsecureSkipVerify bool `json:"insecure_skip_verify,omitempty"` // Disables TLS certificate verification for provider connections - CACertPEM *SecretVar `json:"ca_cert_pem,omitempty"` // PEM-encoded CA certificate to trust for provider endpoint connections (supports env.*) + CACertPEM *SecretVar `json:"ca_cert_pem,omitempty"` // PEM-encoded CA certificate to trust for provider endpoint connections (supports env.*) StreamIdleTimeoutInSeconds int `json:"stream_idle_timeout_in_seconds,omitempty"` // Idle timeout per stream chunk (0 = use default 60s) MaxConnsPerHost int `json:"max_conns_per_host,omitempty"` // Max TCP connections per provider host (default: 5000) EnforceHTTP2 bool `json:"enforce_http2,omitempty"` // Force HTTP/2 on provider connections (relevant for net/http-based providers like Bedrock) @@ -82,7 +82,7 @@ func (nc *NetworkConfig) UnmarshalJSON(data []byte) error { RetryBackoffInitial json.RawMessage `json:"retry_backoff_initial"` // string ("500ms") or int (milliseconds) RetryBackoffMax json.RawMessage `json:"retry_backoff_max"` // string ("5s") or int (milliseconds) InsecureSkipVerify bool `json:"insecure_skip_verify,omitempty"` - CACertPEM *SecretVar `json:"ca_cert_pem,omitempty"` + CACertPEM *SecretVar `json:"ca_cert_pem,omitempty"` StreamIdleTimeoutInSeconds int `json:"stream_idle_timeout_in_seconds,omitempty"` MaxConnsPerHost int `json:"max_conns_per_host,omitempty"` EnforceHTTP2 bool `json:"enforce_http2,omitempty"` @@ -253,11 +253,11 @@ const ( // ProxyConfig holds the configuration for proxy settings. type ProxyConfig struct { - Type ProxyType `json:"type"` // Type of proxy to use - URL *SecretVar `json:"url"` // URL of the proxy server (supports env.*) - Username *SecretVar `json:"username"` // Username for proxy authentication (supports env.*) - Password *SecretVar `json:"password"` // Password for proxy authentication (supports env.*) - CACertPEM *SecretVar `json:"ca_cert_pem"` // PEM-encoded CA certificate to trust for TLS connections through the proxy (supports env.*) + Type ProxyType `json:"type"` // Type of proxy to use + URL *SecretVar `json:"url"` // URL of the proxy server (supports env.*) + Username *SecretVar `json:"username"` // Username for proxy authentication (supports env.*) + Password *SecretVar `json:"password"` // Password for proxy authentication (supports env.*) + CACertPEM *SecretVar `json:"ca_cert_pem"` // PEM-encoded CA certificate to trust for TLS connections through the proxy (supports env.*) } // MarshalForStorage serializes proxy settings for persistence (e.g. proxy_config_json). @@ -322,6 +322,10 @@ type AllowedRequests struct { ChatCompletionStream bool `json:"chat_completion_stream"` Responses bool `json:"responses"` ResponsesStream bool `json:"responses_stream"` + ResponsesRetrieve bool `json:"responses_retrieve"` + ResponsesDelete bool `json:"responses_delete"` + ResponsesCancel bool `json:"responses_cancel"` + ResponsesInputItems bool `json:"responses_input_items"` CountTokens bool `json:"count_tokens"` Compaction bool `json:"compaction"` Embedding bool `json:"embedding"` @@ -394,6 +398,14 @@ func (ar *AllowedRequests) IsOperationAllowed(operation RequestType) bool { return ar.Responses case ResponsesStreamRequest: return ar.ResponsesStream + case ResponsesRetrieveRequest: + return ar.ResponsesRetrieve + case ResponsesDeleteRequest: + return ar.ResponsesDelete + case ResponsesCancelRequest: + return ar.ResponsesCancel + case ResponsesInputItemsRequest: + return ar.ResponsesInputItems case CountTokensRequest: return ar.CountTokens case CompactionRequest: @@ -706,6 +718,16 @@ type Provider interface { PassthroughStream(ctx *BifrostContext, postHookRunner PostHookRunner, postHookSpanFinalizer func(context.Context), key Key, req *BifrostPassthroughRequest) (chan *BifrostStreamChunk, *BifrostError) } +// ResponsesLifecycleProvider is an optional interface for OpenAI-style Responses API +// secondary verbs (retrieve, delete, cancel, list input items). Checked via type assertion +// in core dispatch; providers that do not implement it return unsupported_operation. +type ResponsesLifecycleProvider interface { + ResponsesRetrieve(ctx *BifrostContext, key Key, req *BifrostResponsesRetrieveRequest) (*BifrostResponsesResponse, *BifrostError) + ResponsesDelete(ctx *BifrostContext, key Key, req *BifrostResponsesDeleteRequest) (*BifrostResponsesDeleteResponse, *BifrostError) + ResponsesCancel(ctx *BifrostContext, key Key, req *BifrostResponsesCancelRequest) (*BifrostResponsesResponse, *BifrostError) + ResponsesInputItems(ctx *BifrostContext, key Key, req *BifrostResponsesInputItemsRequest) (*BifrostResponsesInputItemsResponse, *BifrostError) +} + // WebSocketCapableProvider is an optional interface that providers can implement // to indicate support for the OpenAI Responses API WebSocket Mode. // Checked via type assertion: provider.(WebSocketCapableProvider). diff --git a/core/schemas/responses.go b/core/schemas/responses.go index 30428384a33..7119f7b748a 100644 --- a/core/schemas/responses.go +++ b/core/schemas/responses.go @@ -48,6 +48,98 @@ func (r *BifrostResponsesRequest) GetRawRequestBody() []byte { return r.RawRequestBody } +// BifrostResponsesRetrieveRequest retrieves a stored response by ID (OpenAI GET /v1/responses/{id}). +// +// Multi-key note: when multiple API keys are configured for the same provider, pin +// key selection (for example x-bf-api-key-id) on lifecycle calls so they hit the same +// upstream account as the create that produced response_id. +type BifrostResponsesRetrieveRequest struct { + Provider ModelProvider `json:"provider"` + ResponseID string `json:"response_id"` + Include []string `json:"include,omitempty"` + StartingAfter *int `json:"starting_after,omitempty"` + IncludeObfuscation *bool `json:"include_obfuscation,omitempty"` + RawRequestBody []byte `json:"-"` +} + +// GetRawRequestBody implements raw body passthrough when enabled on context. +func (r *BifrostResponsesRetrieveRequest) GetRawRequestBody() []byte { + if r == nil { + return nil + } + return r.RawRequestBody +} + +// BifrostResponsesDeleteRequest deletes a stored response (OpenAI DELETE /v1/responses/{id}). +// See BifrostResponsesRetrieveRequest for multi-key pinning guidance. +type BifrostResponsesDeleteRequest struct { + Provider ModelProvider `json:"provider"` + ResponseID string `json:"response_id"` + RawRequestBody []byte `json:"-"` +} + +// GetRawRequestBody implements raw body passthrough when enabled on context. +func (r *BifrostResponsesDeleteRequest) GetRawRequestBody() []byte { + if r == nil { + return nil + } + return r.RawRequestBody +} + +// BifrostResponsesCancelRequest cancels an in-flight stored response (OpenAI POST /v1/responses/{id}/cancel). +// See BifrostResponsesRetrieveRequest for multi-key pinning guidance. +type BifrostResponsesCancelRequest struct { + Provider ModelProvider `json:"provider"` + ResponseID string `json:"response_id"` + RawRequestBody []byte `json:"-"` +} + +// GetRawRequestBody implements raw body passthrough when enabled on context. +func (r *BifrostResponsesCancelRequest) GetRawRequestBody() []byte { + if r == nil { + return nil + } + return r.RawRequestBody +} + +// BifrostResponsesInputItemsRequest lists input items for a response (OpenAI GET /v1/responses/{id}/input_items). +// See BifrostResponsesRetrieveRequest for multi-key pinning guidance. +type BifrostResponsesInputItemsRequest struct { + Provider ModelProvider `json:"provider"` + ResponseID string `json:"response_id"` + After string `json:"after,omitempty"` + Include []string `json:"include,omitempty"` + Limit *int `json:"limit,omitempty"` + Order string `json:"order,omitempty"` + RawRequestBody []byte `json:"-"` +} + +// GetRawRequestBody implements raw body passthrough when enabled on context. +func (r *BifrostResponsesInputItemsRequest) GetRawRequestBody() []byte { + if r == nil { + return nil + } + return r.RawRequestBody +} + +// BifrostResponsesDeleteResponse is the wire shape for a successful delete of a stored response. +type BifrostResponsesDeleteResponse struct { + ID string `json:"id"` + Object string `json:"object,omitempty"` + Deleted bool `json:"deleted"` + ExtraFields BifrostResponseExtraFields `json:"extra_fields"` +} + +// BifrostResponsesInputItemsResponse is the list payload for response input items. +type BifrostResponsesInputItemsResponse struct { + Object string `json:"object"` + Data []ResponsesMessage `json:"data"` + HasMore bool `json:"has_more"` + FirstID string `json:"first_id,omitempty"` + LastID string `json:"last_id,omitempty"` + ExtraFields BifrostResponseExtraFields `json:"extra_fields"` +} + // BifrostCompactionRequest is the request for the context compaction endpoint (POST /v1/responses/compact). // It is a strict subset of BifrostResponsesRequest — tools, sampling params, and streaming are not supported. type BifrostCompactionRequest struct { @@ -802,6 +894,22 @@ type ResponsesResponseIncompleteDetails struct { Reason string `json:"reason"` // The reason why the response is incomplete } +// ResponsesResponse.Status values (OpenAI Responses API). +const ( + ResponsesResponseStatusInProgress = "in_progress" + ResponsesResponseStatusCompleted = "completed" + ResponsesResponseStatusIncomplete = "incomplete" + ResponsesResponseStatusFailed = "failed" + ResponsesResponseStatusCancelled = "cancelled" + ResponsesResponseStatusQueued = "queued" +) + +// ResponsesResponseIncompleteDetails.Reason values. +const ( + ResponsesResponseIncompleteReasonMaxOutputTokens = "max_output_tokens" + ResponsesResponseIncompleteReasonContentFilter = "content_filter" +) + type ResponsesResponseUsage struct { Type *string `json:"type,omitempty"` // type field is sent by anthropic InputTokens int `json:"input_tokens"` // Number of input tokens (prompt tokens + cached tokens) @@ -874,7 +982,9 @@ func (d *ResponsesResponseInputTokens) UnmarshalJSON(data []byte) error { return nil } -// MarshalJSON emits cached_tokens (read+write) alongside the individual fields for OpenAI spec compatibility. +// MarshalJSON emits cached_tokens (reads only, per the OpenAI spec and mirroring UnmarshalJSON above) alongside the individual fields. +// Cache writes are reported separately via cached_write_tokens and are excluded from cached_tokens so that +// OpenAI-spec consumers do not price cache writes as cache reads. func (d ResponsesResponseInputTokens) MarshalJSON() ([]byte, error) { type raw struct { TextTokens int `json:"text_tokens,omitempty"` @@ -892,7 +1002,7 @@ func (d ResponsesResponseInputTokens) MarshalJSON() ([]byte, error) { CachedReadTokens: d.CachedReadTokens, CachedWriteTokens: d.CachedWriteTokens, CachedWriteTokenDetails: d.CachedWriteTokenDetails, - CachedTokens: d.CachedReadTokens + d.CachedWriteTokens, + CachedTokens: d.CachedReadTokens, }) } @@ -936,7 +1046,14 @@ const ( ResponsesMessageTypeItemReference ResponsesMessageType = "item_reference" ResponsesMessageTypeRefusal ResponsesMessageType = "refusal" ResponsesMessageTypeCompaction ResponsesMessageType = "compaction" - ResponsesMessageTypeAdvisorCall ResponsesMessageType = "advisor_call" // Anthropic advisor server tool (server_tool_use + advisor_tool_result) + // Codex deferred-tool discovery (tool_search). OpenAI's Responses API + // supports these item types natively; Bifrost preserves them verbatim + // because its typed schema doesn't model them (the call's `arguments` is a + // JSON object — unlike function_call's string — and the output carries a + // `tools` array). See ResponsesMessage's (Un)MarshalJSON. + ResponsesMessageTypeToolSearchCall ResponsesMessageType = "tool_search_call" + ResponsesMessageTypeToolSearchOutput ResponsesMessageType = "tool_search_output" + ResponsesMessageTypeAdvisorCall ResponsesMessageType = "advisor_call" // Anthropic advisor server tool (server_tool_use + advisor_tool_result) ) // ResponsesMessage is a union type that can contain different types of input items @@ -965,20 +1082,47 @@ type ResponsesMessage struct { // Reasoning // gpt-oss models include only reasoning_text content blocks in a message, while other openai models include summaries+encrypted_content *ResponsesReasoning + + // rawToolSearch preserves codex `tool_search_call` / `tool_search_output` + // items verbatim. OpenAI's Responses API accepts these natively, but + // Bifrost's typed schema doesn't model them (the call's `arguments` is a + // JSON object — unlike function_call's string — and the output carries a + // `tools` array). Rather than fail to deserialize the whole input array or + // drop/mangle these items, we round-trip the original bytes unchanged. + // Set by UnmarshalJSON, emitted by MarshalJSON; nil for every other type. + rawToolSearch []byte +} + +// isToolSearchItem reports whether t is a codex tool_search item type, which +// Bifrost preserves verbatim rather than modelling field-by-field. +func isToolSearchItem(t string) bool { + return t == string(ResponsesMessageTypeToolSearchCall) || + t == string(ResponsesMessageTypeToolSearchOutput) } -// UnmarshalJSON normalizes function/tool-call arguments before decoding the rest +// UnmarshalJSON preserves codex tool_search items verbatim (see rawToolSearch) +// and otherwise normalizes function/tool-call arguments before decoding the rest // of the item. OpenAI's Responses API serializes `function_call` `arguments` as -// a JSON string, but `tool_search_call` items (emitted when the request enables -// the `tool_search` deferred-tool-discovery tool, as Codex does) serialize -// `arguments` as a JSON object — e.g. {} while in_progress and -// {"query":"...","limit":10} when completed. The embedded -// ResponsesToolMessage.Arguments field is a *string, so an object value makes a -// plain decode fail with "Mismatch type string with value object", which -// silently drops the item mid-stream and hangs streaming clients. We shadow -// `arguments` as raw JSON, decode everything else as usual, then store the -// canonical stringified form. +// a JSON string, but `tool_search_call` items serialize `arguments` as a JSON +// object — e.g. {} while in_progress and {"query":"...","limit":10} when +// completed. The embedded ResponsesToolMessage.Arguments field is a *string, so +// an object value makes a plain decode fail with "Mismatch type string with +// value object", which silently drops the item mid-stream and hangs streaming +// clients. We shadow `arguments` as raw JSON, decode everything else as usual, +// then store the canonical stringified form. func (m *ResponsesMessage) UnmarshalJSON(data []byte) error { + // Clear the receiver first so a reused instance never retains a stale + // rawToolSearch (or other fields) from a prior decode — unmarshalling a + // non-tool-search payload must not leave preserved bytes that MarshalJSON + // would then re-emit. + *m = ResponsesMessage{} + if t := gjson.GetBytes(data, "type").String(); isToolSearchItem(t) { + mt := ResponsesMessageType(t) + m.Type = &mt + m.rawToolSearch = append([]byte(nil), data...) + return nil + } + type Alias ResponsesMessage aux := &struct { Arguments json.RawMessage `json:"arguments,omitempty"` @@ -1002,6 +1146,16 @@ func (m *ResponsesMessage) UnmarshalJSON(data []byte) error { return nil } +// MarshalJSON re-emits preserved tool_search items verbatim and defers every +// other item type to the default (sorted-key) struct encoding. +func (m ResponsesMessage) MarshalJSON() ([]byte, error) { + if m.rawToolSearch != nil { + return m.rawToolSearch, nil + } + type alias ResponsesMessage + return MarshalSorted(alias(m)) +} + // responsesToolArgumentsToString normalizes a function/tool-call `arguments` // value into the stringified-JSON form expected downstream. function_call items // send a JSON string; tool_search_call items send a JSON object. Both are @@ -1210,6 +1364,9 @@ type ResponsesToolMessage struct { // Anthropic advisor-specific (advisor_call): carries the advisor_tool_result payload *ResponsesAdvisorCall + // Anthropic web-fetch-specific (web_fetch_call): carries the web_fetch_tool_result payload + *ResponsesWebFetchCall + // Anthropic code-execution-specific (code_interpreter_call): carries the // server_tool_use input + *_code_execution_tool_result payload that the // neutral ResponsesCodeInterpreterToolCall cannot represent. @@ -1226,6 +1383,34 @@ type ResponsesAdvisorCall struct { StopReason *string `json:"advisor_stop_reason,omitempty"` // present when max_tokens is set on the tool } +// ResponsesWebFetchCall carries the Anthropic web_fetch_tool_result payload +// alongside a web_fetch_call. Anthropic-only; the request URL lives on +// ResponsesWebFetchToolCallAction. +type ResponsesWebFetchCall struct { + ResultType string `json:"web_fetch_result_type,omitempty"` // "web_fetch_result" | "web_fetch_tool_result_error" + URL *string `json:"web_fetch_result_url,omitempty"` + RetrievedAt *string `json:"web_fetch_retrieved_at,omitempty"` + Document *ResponsesWebFetchDocument `json:"web_fetch_document,omitempty"` + ErrorCode *string `json:"web_fetch_error_code,omitempty"` +} + +type ResponsesWebFetchDocument struct { + Type string `json:"type,omitempty"` // "document" + Text *string `json:"text,omitempty"` + Title *string `json:"title,omitempty"` + Source *ResponsesWebFetchSource `json:"source,omitempty"` + Citations *Citations `json:"citations,omitempty"` + Context *string `json:"context,omitempty"` +} + +type ResponsesWebFetchSource struct { + Type string `json:"type,omitempty"` // "text" | "base64" | "url" | "file" + MediaType *string `json:"media_type,omitempty"` + Data *string `json:"data,omitempty"` + URL *string `json:"url,omitempty"` + FileID *string `json:"file_id,omitempty"` +} + // ResponsesToolCaller is the neutral form of Anthropic's "caller" union on // server_tool_use / *_tool_result blocks. It links a tool call to the agentic // caller that produced it (e.g. programmatic tool calling from inside the code @@ -1399,7 +1584,11 @@ func (output ResponsesToolMessageOutputStruct) MarshalJSON() ([]byte, error) { if output.ResponsesComputerToolCallOutput != nil { return MarshalSorted(output.ResponsesComputerToolCallOutput) } - return nil, fmt.Errorf("responses tool message output struct is neither a string nor an array of responses message content blocks nor a computer tool call output data nor an image generation call output") + // All variants nil: a tool legitimately produced no output (e.g. an + // Anthropic tool_result with empty content). Serialize as an empty string + // rather than erroring, since an error here aborts marshaling of any + // enclosing structure (conversation histories, log rows). + return MarshalSorted("") } func (output *ResponsesToolMessageOutputStruct) UnmarshalJSON(data []byte) error { @@ -2728,9 +2917,11 @@ type ResponsesToolToolSearch struct { // ResponsesToolWebFetch represents a web fetch tool type ResponsesToolWebFetch struct { - MaxUses *int `json:"max_uses,omitempty"` - Filters *ResponsesToolWebSearchFilters `json:"filters,omitempty"` - MaxContentTokens *int `json:"max_content_tokens,omitempty"` + MaxUses *int `json:"max_uses,omitempty"` + Filters *ResponsesToolWebSearchFilters `json:"filters,omitempty"` + MaxContentTokens *int `json:"max_content_tokens,omitempty"` + UseCache *bool `json:"use_cache,omitempty"` + ResponseInclusion *string `json:"response_inclusion,omitempty"` // "full" | "excluded" (web_fetch_20260318+) } // ResponsesToolAdvisorCaching toggles advisor-side prompt caching. diff --git a/core/schemas/responses_test.go b/core/schemas/responses_test.go index 6b4f18dcbba..28692507339 100644 --- a/core/schemas/responses_test.go +++ b/core/schemas/responses_test.go @@ -231,26 +231,32 @@ func TestResponsesMessageToolCallArguments(t *testing.T) { // Real tool_search_call frames captured from api.openai.com by replaying // Codex's request (which enables the `tool_search` tool). These are the exact // frames that triggered the production "Mismatch type string with value - // object" failure. + // object" failure. tool_search items are preserved verbatim (see + // rawToolSearch), so the item must decode without error and re-encode + // byte-identically, object-form arguments included. t.Run("real tool_search_call frames from openai", func(t *testing.T) { - frames := map[string]string{ - "in_progress (empty object)": `{"type":"response.output_item.added","output_index":1,"sequence_number":4,"item":{"id":"tsc_01429bcd111d3db1016a3abc8e12948191a9efb0edcbd7f68a","type":"tool_search_call","status":"in_progress","arguments":{},"call_id":"call_OYgDGFxcFL8POxRYssDHUsaM","execution":"client"}}`, - "completed (populated object)": `{"type":"response.output_item.done","output_index":1,"sequence_number":5,"item":{"id":"tsc_01429bcd111d3db1016a3abc8e12948191a9efb0edcbd7f68a","type":"tool_search_call","status":"completed","arguments":{"query":"observability_repro sentry grafana websocket responses","limit":10},"call_id":"call_OYgDGFxcFL8POxRYssDHUsaM","execution":"client"}}`, + items := map[string]string{ + "in_progress (empty object)": `{"id":"tsc_01429bcd111d3db1016a3abc8e12948191a9efb0edcbd7f68a","type":"tool_search_call","status":"in_progress","arguments":{},"call_id":"call_OYgDGFxcFL8POxRYssDHUsaM","execution":"client"}`, + "completed (populated object)": `{"id":"tsc_01429bcd111d3db1016a3abc8e12948191a9efb0edcbd7f68a","type":"tool_search_call","status":"completed","arguments":{"query":"observability_repro sentry grafana websocket responses","limit":10},"call_id":"call_OYgDGFxcFL8POxRYssDHUsaM","execution":"client"}`, } - want := map[string]string{ - "in_progress (empty object)": `{}`, - "completed (populated object)": `{"query":"observability_repro sentry grafana websocket responses","limit":10}`, + events := map[string]string{ + "in_progress (empty object)": `{"type":"response.output_item.added","output_index":1,"sequence_number":4,"item":` + items["in_progress (empty object)"] + `}`, + "completed (populated object)": `{"type":"response.output_item.done","output_index":1,"sequence_number":5,"item":` + items["completed (populated object)"] + `}`, } - for name, raw := range frames { + for name, raw := range events { var resp BifrostResponsesStreamResponse if err := Unmarshal([]byte(raw), &resp); err != nil { t.Fatalf("[%s] unmarshal tool_search_call frame: %v", name, err) } - if resp.Item == nil || resp.Item.ResponsesToolMessage == nil || resp.Item.Arguments == nil { - t.Fatalf("[%s] expected tool_search_call arguments to be set, got %#v", name, resp.Item) + if resp.Item == nil || resp.Item.Type == nil || *resp.Item.Type != ResponsesMessageTypeToolSearchCall { + t.Fatalf("[%s] expected tool_search_call item, got %#v", name, resp.Item) } - if *resp.Item.Arguments != want[name] { - t.Fatalf("[%s] expected arguments %q, got %q", name, want[name], *resp.Item.Arguments) + encoded, err := MarshalSorted(resp.Item) + if err != nil { + t.Fatalf("[%s] marshal preserved tool_search_call item: %v", name, err) + } + if string(encoded) != items[name] { + t.Fatalf("[%s] expected item to round-trip verbatim\nwant: %s\ngot: %s", name, items[name], encoded) } } }) diff --git a/core/schemas/responsestooloutput_test.go b/core/schemas/responsestooloutput_test.go new file mode 100644 index 00000000000..8ffe8862e02 --- /dev/null +++ b/core/schemas/responsestooloutput_test.go @@ -0,0 +1,55 @@ +package schemas + +import ( + "testing" +) + +// TestResponsesToolMessageOutputStructMarshalEmpty verifies that an output +// struct with all variants nil serializes as an empty string instead of +// erroring. An error here would abort marshaling of any enclosing structure +// (conversation histories, log rows), silently dropping data downstream. +func TestResponsesToolMessageOutputStructMarshalEmpty(t *testing.T) { + data, err := MarshalSorted(ResponsesToolMessageOutputStruct{}) + if err != nil { + t.Fatalf("empty output struct must marshal, got error: %v", err) + } + if string(data) != `""` { + t.Fatalf("empty output struct should marshal as empty string, got %s", data) + } +} + +// TestResponsesToolMessageOutputStructRoundTripEmpty verifies the marshaled +// empty output unmarshals back into the string variant. +func TestResponsesToolMessageOutputStructRoundTripEmpty(t *testing.T) { + data, err := MarshalSorted(ResponsesToolMessageOutputStruct{}) + if err != nil { + t.Fatalf("marshal: %v", err) + } + var out ResponsesToolMessageOutputStruct + if err := Unmarshal(data, &out); err != nil { + t.Fatalf("unmarshal: %v", err) + } + if out.ResponsesToolCallOutputStr == nil || *out.ResponsesToolCallOutputStr != "" { + t.Fatalf("round trip should yield empty output string, got %+v", out) + } +} + +// TestResponsesMessageMarshalWithEmptyToolOutput verifies a full message +// containing an empty tool output (the shape produced by an Anthropic +// tool_result with content: []) serializes cleanly inside a slice, matching +// how conversation histories are stored. +func TestResponsesMessageMarshalWithEmptyToolOutput(t *testing.T) { + msgs := []ResponsesMessage{ + { + Type: Ptr(ResponsesMessageTypeFunctionCallOutput), + Status: Ptr("completed"), + ResponsesToolMessage: &ResponsesToolMessage{ + CallID: Ptr("toolu_empty"), + Output: &ResponsesToolMessageOutputStruct{}, + }, + }, + } + if _, err := MarshalSorted(msgs); err != nil { + t.Fatalf("history containing empty tool output must marshal, got: %v", err) + } +} diff --git a/core/schemas/secretvar.go b/core/schemas/secretvar.go index 09b946565ae..08c27d0f877 100644 --- a/core/schemas/secretvar.go +++ b/core/schemas/secretvar.go @@ -39,13 +39,15 @@ func inferSecretType(ref string) SecretType { return SecretTypePlainText } -// NewSecretVar creates a new SecretVar from a string. -func NewSecretVar(value string) *SecretVar { +// parseSecretRef classifies value and returns a *SecretVar with SecretType and ref +// populated but Val unresolved — no vault HTTP calls, no env lookups. +// For JSON-encoded SecretVars the raw "value" field is preserved in Val unchanged. +// Adding a new secret type only requires updating this function. +func parseSecretRef(value string) *SecretVar { val := value if unquoted, err := strconv.Unquote(value); err == nil { val = unquoted } - // If it's a valid JSON object following the SecretVar schema, unmarshal it if sonic.Valid([]byte(value)) { valueNode, _ := sonic.Get([]byte(val), "value") if valueNode.Exists() { @@ -76,55 +78,52 @@ func NewSecretVar(value string) *SecretVar { } e.ref = ref e.SecretType = SecretTypeEnv - if envValue, ok := os.LookupEnv(strings.TrimPrefix(ref, "env.")); ok { - e.Val = envValue - } else { - e.Val = "" - } - return e } else if strings.HasPrefix(raw.Val, "env.") && raw.Val == raw.EnvVar { // Legacy format: value == env_var == "env.XXX" e.ref = raw.EnvVar e.SecretType = SecretTypeEnv - e.Val = "" - if envValue, ok := os.LookupEnv(strings.TrimPrefix(raw.EnvVar, "env.")); ok { - e.Val = envValue - } - return e - } - // Resolve references - if e.SecretType == SecretTypeVault { - e.Val = "" - if vaultValue, ok := LookupVault(e.ref); ok { - e.Val = vaultValue - } - } - if e.SecretType == SecretTypeEnv { - if envValue, ok := os.LookupEnv(e.EnvKey()); ok { - e.Val = envValue - } else { - e.Val = "" - } + } else { + // Plain text JSON object ({value, ...} with no type/ref/from_env). + e.SecretType = SecretTypePlainText } return e } } } if strings.HasPrefix(val, "vault.") { - e := &SecretVar{ref: val, SecretType: SecretTypeVault} - if vaultValue, ok := LookupVault(val); ok { + return &SecretVar{ref: val, SecretType: SecretTypeVault} + } + if strings.HasPrefix(val, "env.") { + return &SecretVar{ref: val, SecretType: SecretTypeEnv} + } + return &SecretVar{Val: val, SecretType: SecretTypePlainText} +} + +// IsSecretRef reports whether value is a secret reference (env.* or vault.* prefix, +// or a JSON-encoded SecretVar with an env/vault type) without resolving it. +// Use this instead of NewSecretVar(...).IsFromSecret() when resolution side-effects +// (vault HTTP calls, env lookups) must be avoided. +func IsSecretRef(value string) bool { + return parseSecretRef(value).IsFromSecret() +} + +// NewSecretVar creates a new SecretVar from a string. +func NewSecretVar(value string) *SecretVar { + e := parseSecretRef(value) + switch e.SecretType { + case SecretTypeVault: + e.Val = "" + if vaultValue, ok := LookupVault(e.ref); ok { e.Val = vaultValue } - return e - } - if envKey, ok := strings.CutPrefix(val, "env."); ok { - e := &SecretVar{ref: val, SecretType: SecretTypeEnv} - if envValue, ok := os.LookupEnv(envKey); ok { + case SecretTypeEnv: + if envValue, ok := os.LookupEnv(e.EnvKey()); ok { e.Val = envValue + } else { + e.Val = "" } - return e } - return &SecretVar{Val: val} + return e } // GetRawRef returns the full secret reference string including prefix @@ -358,6 +357,9 @@ func (e *SecretVar) UnmarshalJSON(data []byte) error { e.Val = "" } } + if e.SecretType == "" { + e.SecretType = SecretTypePlainText + } return nil } } diff --git a/core/schemas/serialization_test.go b/core/schemas/serialization_test.go index 1c91b0b94ed..e46bc8b7117 100644 --- a/core/schemas/serialization_test.go +++ b/core/schemas/serialization_test.go @@ -1545,3 +1545,78 @@ func TestSonic_ToolFunctionParameters_DeepCopy_KeyOrderIndependent(t *testing.T) assert.NotEqual(t, original.keyOrder.keys[0], copied.keyOrder.keys[0], "copy must not share JSONKeyOrder.keys backing array") assert.Equal(t, "$defs", copied.keyOrder.keys[0]) } + +// --- ChatPromptTokensDetails / ResponsesResponseInputTokens cached_tokens --- +// Per the OpenAI spec, cached_tokens counts prompt tokens read from the cache. Cache +// writes must not be folded in, or spec consumers price cache writes as cache reads. + +func TestSonic_ChatPromptTokensDetails_CachedTokensExcludesWrites(t *testing.T) { + // Fresh-cache turn: write only, no read. cached_tokens must stay 0. + out, err := Marshal(ChatPromptTokensDetails{CachedWriteTokens: 9106}) + require.NoError(t, err) + var m map[string]any + require.NoError(t, json.Unmarshal(out, &m)) + assert.Equal(t, float64(0), m["cached_tokens"]) + assert.Equal(t, float64(9106), m["cached_write_tokens"]) + + // Cache-hit turn with a concurrent write: cached_tokens must equal reads only. + out, err = Marshal(ChatPromptTokensDetails{CachedReadTokens: 500, CachedWriteTokens: 100}) + require.NoError(t, err) + m = nil + require.NoError(t, json.Unmarshal(out, &m)) + assert.Equal(t, float64(500), m["cached_tokens"]) + assert.Equal(t, float64(500), m["cached_read_tokens"]) + assert.Equal(t, float64(100), m["cached_write_tokens"]) +} + +func TestSonic_ChatPromptTokensDetails_CachedTokensRoundTrip(t *testing.T) { + // Round-trip must preserve the read/write split; the bare cached_tokens fallback + // in UnmarshalJSON only applies when the split fields are absent. + in := ChatPromptTokensDetails{CachedReadTokens: 500, CachedWriteTokens: 100} + out, err := Marshal(in) + require.NoError(t, err) + var back ChatPromptTokensDetails + require.NoError(t, Unmarshal(out, &back)) + assert.Equal(t, in.CachedReadTokens, back.CachedReadTokens) + assert.Equal(t, in.CachedWriteTokens, back.CachedWriteTokens) + + // OpenAI-spec providers send only cached_tokens; it maps to reads. + var d ChatPromptTokensDetails + require.NoError(t, Unmarshal([]byte(`{"cached_tokens":42}`), &d)) + assert.Equal(t, 42, d.CachedReadTokens) + assert.Equal(t, 0, d.CachedWriteTokens) +} + +func TestSonic_ResponsesResponseInputTokens_CachedTokensExcludesWrites(t *testing.T) { + // Fresh-cache turn: write only, no read. cached_tokens must stay 0. + out, err := Marshal(ResponsesResponseInputTokens{CachedWriteTokens: 9106}) + require.NoError(t, err) + var m map[string]any + require.NoError(t, json.Unmarshal(out, &m)) + assert.Equal(t, float64(0), m["cached_tokens"]) + assert.Equal(t, float64(9106), m["cached_write_tokens"]) + + // Cache-hit turn with a concurrent write: cached_tokens must equal reads only. + out, err = Marshal(ResponsesResponseInputTokens{CachedReadTokens: 500, CachedWriteTokens: 100}) + require.NoError(t, err) + m = nil + require.NoError(t, json.Unmarshal(out, &m)) + assert.Equal(t, float64(500), m["cached_tokens"]) + assert.Equal(t, float64(500), m["cached_read_tokens"]) + assert.Equal(t, float64(100), m["cached_write_tokens"]) +} + +func TestSonic_ResponsesResponseInputTokens_CachedTokensRoundTrip(t *testing.T) { + in := ResponsesResponseInputTokens{CachedReadTokens: 500, CachedWriteTokens: 100} + out, err := Marshal(in) + require.NoError(t, err) + var back ResponsesResponseInputTokens + require.NoError(t, Unmarshal(out, &back)) + assert.Equal(t, in.CachedReadTokens, back.CachedReadTokens) + assert.Equal(t, in.CachedWriteTokens, back.CachedWriteTokens) + + var d ResponsesResponseInputTokens + require.NoError(t, Unmarshal([]byte(`{"cached_tokens":42}`), &d)) + assert.Equal(t, 42, d.CachedReadTokens) + assert.Equal(t, 0, d.CachedWriteTokens) +} diff --git a/core/schemas/trace.go b/core/schemas/trace.go index 0bb9bf6f996..d3b6cb0e439 100644 --- a/core/schemas/trace.go +++ b/core/schemas/trace.go @@ -493,6 +493,12 @@ const ( AttrBifrostCustomerName = "bifrost.customer.name" AttrBifrostBusinessUnitID = "bifrost.business_unit.id" AttrBifrostBusinessUnitName = "bifrost.business_unit.name" + AttrBifrostTeamIDs = "bifrost.team.ids" + AttrBifrostTeamNames = "bifrost.team.names" + AttrBifrostCustomerIDs = "bifrost.customer.ids" + AttrBifrostCustomerNames = "bifrost.customer.names" + AttrBifrostBusinessUnitIDs = "bifrost.business_unit.ids" + AttrBifrostBusinessUnitNames = "bifrost.business_unit.names" AttrBifrostUserID = "bifrost.user.id" AttrBifrostUserName = "bifrost.user.name" AttrBifrostRetries = "bifrost.retries" diff --git a/core/schemas/utils.go b/core/schemas/utils.go index 838190a5304..66cb8c07304 100644 --- a/core/schemas/utils.go +++ b/core/schemas/utils.go @@ -174,10 +174,28 @@ type URLTypeInfo struct { DataURLWithoutPrefix *string // URL without the prefix (eg data:image/png;base64,iVBORw0KGgo...) } -// SanitizeImageURL sanitizes and validates an image URL. -// It handles both data URLs and regular HTTP/HTTPS URLs. -// It also detects raw base64 image data and adds proper data URL headers. +// defaultImageURLSchemes is the historical allowlist enforced by SanitizeImageURL. +// Provider-specific code paths that need to accept additional schemes (e.g. Vertex's +// gs://) must use SanitizeImageURLWithAllowedSchemes with an explicit list. +var defaultImageURLSchemes = []string{"http", "https"} + +// SanitizeImageURL sanitizes and normalizes an image URL. +// It handles both data URLs and regular URLs, accepting only http/https for non-data +// URLs. Callers that need to accept provider-specific schemes (e.g. gs://) must use +// SanitizeImageURLWithAllowedSchemes. func SanitizeImageURL(rawURL string) (string, error) { + return sanitizeImageURL(rawURL, defaultImageURLSchemes) +} + +// SanitizeImageURLWithAllowedSchemes sanitizes and normalizes an image URL, then +// validates regular URL schemes against the target provider's allowlist. Passing an +// empty list is treated as "no non-data URL is acceptable" — call SanitizeImageURL +// if you want the default http/https policy. +func SanitizeImageURLWithAllowedSchemes(rawURL string, allowedSchemes ...string) (string, error) { + return sanitizeImageURL(rawURL, allowedSchemes) +} + +func sanitizeImageURL(rawURL string, allowedSchemes []string) (string, error) { if rawURL == "" { return rawURL, fmt.Errorf("URL cannot be empty") } @@ -212,9 +230,22 @@ func SanitizeImageURL(rawURL string) (string, error) { return rawURL, fmt.Errorf("invalid URL format: %w", err) } - // Validate scheme - if parsedURL.Scheme != "http" && parsedURL.Scheme != "https" { - return rawURL, fmt.Errorf("URL must use http or https scheme") + if parsedURL.Scheme == "" { + return rawURL, fmt.Errorf("URL must have a valid scheme") + } + + if len(allowedSchemes) == 0 { + return rawURL, fmt.Errorf("URL scheme %q is not allowed: no schemes permitted", parsedURL.Scheme) + } + allowed := false + for _, scheme := range allowedSchemes { + if strings.EqualFold(parsedURL.Scheme, scheme) { + allowed = true + break + } + } + if !allowed { + return rawURL, fmt.Errorf("URL scheme %q is not allowed; expected one of: %s", parsedURL.Scheme, strings.Join(allowedSchemes, ", ")) } // Validate host @@ -1350,6 +1381,15 @@ func DeepCopyResponsesMessage(original ResponsesMessage) ResponsesMessage { } } + if original.ResponsesToolMessage.Caller != nil { + copyCaller := *original.ResponsesToolMessage.Caller + if original.ResponsesToolMessage.Caller.ToolID != nil { + copyToolID := *original.ResponsesToolMessage.Caller.ToolID + copyCaller.ToolID = ©ToolID + } + copy.ResponsesToolMessage.Caller = ©Caller + } + // Deep copy embedded tool call structs (simplified version - add more as needed) if original.ResponsesToolMessage.ResponsesFileSearchToolCall != nil { copyToolCall := *original.ResponsesToolMessage.ResponsesFileSearchToolCall @@ -1363,6 +1403,23 @@ func DeepCopyResponsesMessage(original ResponsesMessage) ResponsesMessage { copy.ResponsesToolMessage.ResponsesFileSearchToolCall = ©ToolCall } + if original.ResponsesToolMessage.ResponsesWebFetchCall != nil { + copyCall := *original.ResponsesToolMessage.ResponsesWebFetchCall + if original.ResponsesToolMessage.ResponsesWebFetchCall.Document != nil { + docCopy := *original.ResponsesToolMessage.ResponsesWebFetchCall.Document + if original.ResponsesToolMessage.ResponsesWebFetchCall.Document.Source != nil { + srcCopy := *original.ResponsesToolMessage.ResponsesWebFetchCall.Document.Source + docCopy.Source = &srcCopy + } + if original.ResponsesToolMessage.ResponsesWebFetchCall.Document.Citations != nil { + citationsCopy := *original.ResponsesToolMessage.ResponsesWebFetchCall.Document.Citations + docCopy.Citations = &citationsCopy + } + copyCall.Document = &docCopy + } + copy.ResponsesToolMessage.ResponsesWebFetchCall = ©Call + } + // Add other embedded tool calls as needed... } @@ -1565,6 +1622,33 @@ func IsAnthropicModel(model string) bool { return strings.Contains(model, "anthropic.") || strings.Contains(model, "claude") } +// IsOpenAIModel checks if the model is an OpenAI model. +func IsOpenAIModel(model string) bool { + if strings.Contains(model, "gpt-") || strings.Contains(model, "text-embedding-") { + return true + } + // OpenAI reasoning families (o1, o3, o4, ...). Match the bare id or a + // version-suffixed variant (e.g. "o3", "o4-mini", "o1-preview") while + // avoiding false matches on substrings like "co1" or "model-o3x". + return isOpenAIReasoningModel(model) +} + +// isOpenAIReasoningModel reports whether model names an OpenAI o-series +// reasoning model. It strips any provider prefix (e.g. "openai/o3") and matches +// an "o" followed by a single digit, where the next character is either end of +// string or a "-" separator, so "o3" and "o4-mini" match but "co1" and "o3x" +// do not. +func isOpenAIReasoningModel(model string) bool { + name := model + if idx := strings.LastIndexAny(name, "/:"); idx >= 0 { + name = name[idx+1:] + } + if len(name) < 2 || name[0] != 'o' || name[1] < '0' || name[1] > '9' { + return false + } + return len(name) == 2 || name[2] == '-' +} + // BedrockModelSupportsCachePoints reports whether the Bedrock model supports // explicit prompt-caching cache points in the Converse API request. func BedrockModelSupportsCachePoints(model string) bool { @@ -1625,6 +1709,11 @@ func IsTitanModel(model string) bool { return strings.Contains(model, "titan") } +// IsGrokModel checks if the model is an xAI Grok model. +func IsGrokModel(model string) bool { + return strings.Contains(model, "grok") +} + // List of grok reasoning models var grokReasoningModels = []string{ "grok-3", diff --git a/core/schemas/utils_test.go b/core/schemas/utils_test.go new file mode 100644 index 00000000000..56db25eb333 --- /dev/null +++ b/core/schemas/utils_test.go @@ -0,0 +1,52 @@ +package schemas + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func TestSanitizeImageURLDefaultRejectsNonHTTPSchemes(t *testing.T) { + // The no-args overload must keep the historical http/https-only policy. Providers + // that legitimately accept other schemes (gs://, file://, ...) must opt in via + // SanitizeImageURLWithAllowedSchemes — otherwise a future caller silently inherits + // a wider attack/regression surface. + _, err := SanitizeImageURL("gs://my-bucket/path/image.png") + require.Error(t, err) + assert.Contains(t, err.Error(), `URL scheme "gs" is not allowed`) + + _, err = SanitizeImageURL("file:///etc/passwd") + require.Error(t, err) +} + +func TestSanitizeImageURLWithAllowedSchemesAcceptsOptIn(t *testing.T) { + sanitizedURL, err := SanitizeImageURLWithAllowedSchemes(" gs://my-bucket/path/image.png ", "http", "https", "gs") + require.NoError(t, err) + assert.Equal(t, "gs://my-bucket/path/image.png", sanitizedURL) +} + +func TestSanitizeImageURLWithAllowedSchemesRejectsUnlisted(t *testing.T) { + _, err := SanitizeImageURLWithAllowedSchemes("gs://my-bucket/path/image.png", "http", "https") + require.Error(t, err) + assert.Contains(t, err.Error(), `URL scheme "gs" is not allowed`) +} + +func TestSanitizeImageURLWithEmptyAllowlistRejects(t *testing.T) { + // Empty allowlist means "no non-data URL is acceptable" — an explicit denial, + // not "fall back to defaults". + _, err := SanitizeImageURLWithAllowedSchemes("https://example.com/foo.png") + require.Error(t, err) + assert.Contains(t, err.Error(), `no schemes permitted`) +} + +func TestSanitizeImageURLDataURLUnaffectedByAllowlist(t *testing.T) { + dataURL := "data:image/png;base64,iVBORw0KGgo=" + got, err := SanitizeImageURL(dataURL) + require.NoError(t, err) + assert.Equal(t, dataURL, got) + + got, err = SanitizeImageURLWithAllowedSchemes(dataURL) + require.NoError(t, err) + assert.Equal(t, dataURL, got) +} diff --git a/core/schemas/vault.go b/core/schemas/vault.go index 5d2868f6821..4fd6f62ba7e 100644 --- a/core/schemas/vault.go +++ b/core/schemas/vault.go @@ -80,25 +80,30 @@ var ( // RemoveOwnedVaultSecretVars best-effort deletes the vault secret for every // SecretVar / *SecretVar field in model whose VaultRef starts with // ownedPrefix+"/". Refs outside that prefix are user-provided and are left alone. -func RemoveOwnedVaultSecretVars(ctx context.Context, ownedPrefix string, model interface{}) { +// Returns one error per field whose deletion failed; callers should log these but +// must not treat them as fatal (the DB row is already deleted). +func RemoveOwnedVaultSecretVars(ctx context.Context, ownedPrefix string, model interface{}) []error { if VaultRemoveHook == nil { - return + return nil } rv := reflect.ValueOf(model) if rv.Kind() == reflect.Ptr { rv = rv.Elem() } if rv.Kind() != reflect.Struct { - return + return nil } rt := rv.Type() + var errs []error for i := 0; i < rt.NumField(); i++ { fv := rv.Field(i) if fv.Type() == secretVarMapType { iter := fv.MapRange() for iter.Next() { e := iter.Value().Interface().(SecretVar) - removeOwnedVaultSecretVar(ctx, ownedPrefix, &e) + if err := removeOwnedVaultSecretVar(ctx, ownedPrefix, &e); err != nil { + errs = append(errs, err) + } } continue } @@ -111,25 +116,28 @@ func RemoveOwnedVaultSecretVars(ctx context.Context, ownedPrefix string, model i field = fv.Interface().(*SecretVar) } } - removeOwnedVaultSecretVar(ctx, ownedPrefix, field) + if err := removeOwnedVaultSecretVar(ctx, ownedPrefix, field); err != nil { + errs = append(errs, err) + } } + return errs } // removeOwnedVaultSecretVar removes a single SecretVar's vault secret if it is a // vault-backed, non-fragment reference under ownedPrefix. Fragment refs (#key) // point at shared, externally-managed secrets and are never auto-deleted. -func removeOwnedVaultSecretVar(ctx context.Context, ownedPrefix string, field *SecretVar) { +func removeOwnedVaultSecretVar(ctx context.Context, ownedPrefix string, field *SecretVar) error { path := field.GetRef() if path == "" { - return + return nil } if strings.IndexByte(path, '#') >= 0 { - return + return nil } if !strings.HasPrefix(path, ownedPrefix+"/") { - return + return nil } - _ = VaultRemoveHook(ctx, path) + return VaultRemoveHook(ctx, path) } // StoreVaultSecretVar pushes a single plaintext SecretVar value into the vault at path diff --git a/core/streamfallback_test.go b/core/streamfallback_test.go new file mode 100644 index 00000000000..175a9305f9c --- /dev/null +++ b/core/streamfallback_test.go @@ -0,0 +1,194 @@ +package bifrost + +import ( + "context" + "fmt" + "net/http" + "net/http/httptest" + "strings" + "sync/atomic" + "testing" + "time" + + schemas "github.com/maximhq/bifrost/core/schemas" +) + +// Regression tests for https://github.com/maximhq/bifrost/issues/4788. +// +// When a streaming attempt fails through an error embedded in an HTTP 200 SSE +// stream (e.g. rate limits sent as SSE events), the provider goroutine exits +// and its teardown claims the connection_closed flag on the request's shared +// BifrostContext (ReleaseStreamingResponse). That claim is scoped to the +// response it released, but the flag stayed set on the context, so the +// idle-timeout reader of the next attempt's stream saw the context as already +// closed and failed every read with "stream closed". Any streaming retry or +// fallback that followed a first-chunk error was dead on arrival. + +// sseHandler serves the given payloads as one SSE data event each. +func sseHandler(payloads ...string) http.HandlerFunc { + return func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "text/event-stream") + fl, _ := w.(http.Flusher) + for _, p := range payloads { + fmt.Fprintf(w, "data: %s\n\n", p) + if fl != nil { + fl.Flush() + } + } + fmt.Fprint(w, "data: [DONE]\n\n") + if fl != nil { + fl.Flush() + } + } +} + +// anthropicMessagesHandler serves a minimal valid Anthropic Messages API +// stream that produces the text "hello". +func anthropicMessagesHandler() http.HandlerFunc { + events := []struct{ typ, data string }{ + {"message_start", `{"type":"message_start","message":{"id":"msg_1","type":"message","role":"assistant","content":[],"model":"claude-3-5-haiku-20241022","usage":{"input_tokens":10,"output_tokens":1}}}`}, + {"content_block_start", `{"type":"content_block_start","index":0,"content_block":{"type":"text","text":""}}`}, + {"content_block_delta", `{"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"hello"}}`}, + {"content_block_stop", `{"type":"content_block_stop","index":0}`}, + {"message_delta", `{"type":"message_delta","delta":{"stop_reason":"end_turn"},"usage":{"output_tokens":2}}`}, + {"message_stop", `{"type":"message_stop"}`}, + } + return func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "text/event-stream") + fl, _ := w.(http.Flusher) + for _, e := range events { + fmt.Fprintf(w, "event: %s\ndata: %s\n\n", e.typ, e.data) + if fl != nil { + fl.Flush() + } + } + } +} + +// drainChatStream collects streamed content and any error chunks. +func drainChatStream(ch chan *schemas.BifrostStreamChunk) (string, []string) { + var content strings.Builder + var errs []string + for chunk := range ch { + if chunk.BifrostError != nil && chunk.BifrostError.Error != nil { + errs = append(errs, chunk.BifrostError.Error.Message) + continue + } + if chunk.BifrostChatResponse == nil { + continue + } + for _, choice := range chunk.BifrostChatResponse.Choices { + if choice.ChatStreamResponseChoice != nil && choice.ChatStreamResponseChoice.Delta != nil && choice.ChatStreamResponseChoice.Delta.Content != nil { + content.WriteString(*choice.ChatStreamResponseChoice.Delta.Content) + } + } + } + return content.String(), errs +} + +func newStreamTestClient(t *testing.T, account *MockAccount) *Bifrost { + t.Helper() + client, err := Init(context.Background(), schemas.BifrostConfig{ + Account: account, + Logger: NewDefaultLogger(schemas.LogLevelError), + }) + if err != nil { + t.Fatalf("failed to initialize bifrost: %v", err) + } + t.Cleanup(client.Shutdown) + return client +} + +func TestStreamFallbackAfterFirstChunkError(t *testing.T) { + primary := httptest.NewServer(sseHandler(`{"error":{"message":"rate limited","type":"rate_limit_error"}}`)) + defer primary.Close() + var fallbackHits atomic.Int32 + fallback := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + fallbackHits.Add(1) + anthropicMessagesHandler()(w, r) + })) + defer fallback.Close() + + account := NewMockAccount() + account.AddProviderWithBaseURL(schemas.OpenAI, 1, 1, primary.URL) + account.AddProviderWithBaseURL(schemas.Anthropic, 1, 1, fallback.URL) + account.configs[schemas.OpenAI].NetworkConfig.MaxRetries = 0 + account.configs[schemas.Anthropic].NetworkConfig.MaxRetries = 0 + account.SetKeysForProvider(schemas.OpenAI, []schemas.Key{ + {ID: "primary-key", Value: *schemas.NewSecretVar("sk-primary"), Models: schemas.WhiteList{"*"}, Weight: 100}, + }) + account.SetKeysForProvider(schemas.Anthropic, []schemas.Key{ + {ID: "fallback-key", Value: *schemas.NewSecretVar("sk-fallback"), Models: schemas.WhiteList{"*"}, Weight: 100}, + }) + client := newStreamTestClient(t, account) + + ctx := schemas.NewBifrostContext(context.Background(), time.Now().Add(30*time.Second)) + stream, bifrostErr := client.ChatCompletionStreamRequest(ctx, &schemas.BifrostChatRequest{ + Provider: schemas.OpenAI, + Model: "gpt-4o-mini", + Input: []schemas.ChatMessage{ + {Role: schemas.ChatMessageRoleUser, Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("hi")}}, + }, + Fallbacks: []schemas.Fallback{{Provider: schemas.Anthropic, Model: "claude-3-5-haiku-20241022"}}, + }) + if bifrostErr != nil { + t.Fatalf("fallback stream failed (fallback server hit %d time(s)): %s", fallbackHits.Load(), bifrostErr.Error.Message) + } + content, errs := drainChatStream(stream) + if got := fallbackHits.Load(); got != 1 { + t.Fatalf("fallback server hits = %d, want 1", got) + } + if len(errs) > 0 { + t.Fatalf("fallback stream emitted error chunks: %v", errs) + } + if content != "hello" { + t.Fatalf("fallback stream content = %q, want %q", content, "hello") + } +} + +func TestStreamRetryAfterFirstChunkError(t *testing.T) { + var hits atomic.Int32 + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if hits.Add(1) == 1 { + sseHandler(`{"error":{"message":"rate limit exceeded, please retry","type":"rate_limit_error"}}`)(w, r) + return + } + sseHandler( + `{"id":"c1","object":"chat.completion.chunk","choices":[{"index":0,"delta":{"role":"assistant","content":"he"}}]}`, + `{"id":"c1","object":"chat.completion.chunk","choices":[{"index":0,"delta":{"content":"llo"}}]}`, + `{"id":"c1","object":"chat.completion.chunk","choices":[{"index":0,"delta":{},"finish_reason":"stop"}],"usage":{"prompt_tokens":10,"completion_tokens":2,"total_tokens":12}}`, + )(w, r) + })) + defer server.Close() + + account := NewMockAccount() + account.AddProviderWithBaseURL(schemas.OpenAI, 1, 1, server.URL) + account.configs[schemas.OpenAI].NetworkConfig.MaxRetries = 1 + account.configs[schemas.OpenAI].NetworkConfig.RetryBackoffInitial = time.Millisecond + account.SetKeysForProvider(schemas.OpenAI, []schemas.Key{ + {ID: "retry-key", Value: *schemas.NewSecretVar("sk-retry"), Models: schemas.WhiteList{"*"}, Weight: 100}, + }) + client := newStreamTestClient(t, account) + + ctx := schemas.NewBifrostContext(context.Background(), time.Now().Add(30*time.Second)) + stream, bifrostErr := client.ChatCompletionStreamRequest(ctx, &schemas.BifrostChatRequest{ + Provider: schemas.OpenAI, + Model: "gpt-4o-mini", + Input: []schemas.ChatMessage{ + {Role: schemas.ChatMessageRoleUser, Content: &schemas.ChatMessageContent{ContentStr: schemas.Ptr("hi")}}, + }, + }) + if bifrostErr != nil { + t.Fatalf("retried stream failed (server hit %d time(s)): %s", hits.Load(), bifrostErr.Error.Message) + } + content, errs := drainChatStream(stream) + if hits.Load() != 2 { + t.Fatalf("server hits = %d, want 2 (initial attempt plus one retry)", hits.Load()) + } + if len(errs) > 0 { + t.Fatalf("retried stream emitted error chunks: %v", errs) + } + if content != "hello" { + t.Fatalf("retried stream content = %q, want %q", content, "hello") + } +} diff --git a/core/utils.go b/core/utils.go index 76cce666345..c5cb180d78c 100644 --- a/core/utils.go +++ b/core/utils.go @@ -83,8 +83,10 @@ var dynamicallyConfigurableProviders = []schemas.ModelProvider{ schemas.Anthropic, schemas.Azure, schemas.Bedrock, + schemas.BedrockMantle, schemas.Cerebras, schemas.Cohere, + schemas.DeepSeek, schemas.Elevenlabs, schemas.Gemini, schemas.Groq, @@ -122,11 +124,11 @@ func providerRequiresKey(customConfig *schemas.CustomProviderConfig) bool { // Some providers like Vertex and Bedrock have their credentials in additional key configs. // Ollama and SGL are keyless (API Key is optional) but use per-key server URLs. func CanProviderKeyValueBeEmpty(providerKey schemas.ModelProvider) bool { - return providerKey == schemas.Vertex || providerKey == schemas.Bedrock || providerKey == schemas.VLLM || providerKey == schemas.Azure || providerKey == schemas.Ollama || providerKey == schemas.SGL + return providerKey == schemas.Vertex || providerKey == schemas.Bedrock || providerKey == schemas.BedrockMantle || providerKey == schemas.VLLM || providerKey == schemas.Azure || providerKey == schemas.Ollama || providerKey == schemas.SGL } func isKeySkippingAllowed(providerKey schemas.ModelProvider) bool { - return providerKey != schemas.Azure && providerKey != schemas.Bedrock && providerKey != schemas.Vertex + return providerKey != schemas.Azure && providerKey != schemas.Bedrock && providerKey != schemas.BedrockMantle && providerKey != schemas.Vertex } // calculateBackoff implements exponential backoff with jitter for retry attempts. @@ -171,6 +173,11 @@ func validateKey(providerKey schemas.ModelProvider, key *schemas.Key) error { if key.BedrockKeyConfig == nil { key.BedrockKeyConfig = &schemas.BedrockKeyConfig{} } + case schemas.BedrockMantle: + // BedrockMantleKeyConfig is optional — an empty config is valid for IRSA / ambient credential auth. + if key.BedrockMantleKeyConfig == nil { + key.BedrockMantleKeyConfig = &schemas.BedrockMantleKeyConfig{} + } case schemas.Vertex: if key.VertexKeyConfig == nil { return fmt.Errorf("vertex_key_config is required") @@ -324,9 +331,53 @@ func clearCtxForFallback(ctx *schemas.BifrostContext) { ctx.ClearValue(schemas.BifrostContextKeyChangeRequestType) ctx.ClearValue(schemas.BifrostContextKeyAttemptTrail) ctx.ClearValue(schemas.BifrostContextKeyStreamEndIndicator) + ctx.ClearValue(schemas.BifrostContextKeyConnectionClosed) ctx.ClearValue(schemas.BifrostContextKeySupportsAssistantPrefill) } +// ClearContextForInternalRequest clears context state that is specific to the +// caller's original request, so a context derived from it can carry an +// internal sub-request (e.g. a plugin generating an embedding for its own +// use) that must behave like a fresh top-level request. +// +// Two categories are cleared: +// +// - Key routing: key-selection state resolved for the caller's provider +// (governance key allow-list, pinned/direct keys, key-selection skip). An +// internal request typically targets a different provider, and when it +// skips the plugin pipeline this state is never re-resolved — inherited, +// it is applied against the wrong provider's key pool and rejects every +// key ("no keys found for provider"). +// - Body transport: raw-body passthrough and large-payload/large-response +// streaming state, plus caller-forwarded extra headers and the caller's +// URL-path override. Inherited, these make providers send the caller's +// raw or streamed body instead of marshaling the internal request, route +// it to the caller's endpoint path instead of the internal request's own, +// and forward the caller's headers on a call the caller doesn't own. +// +// Deliberately not cleared: tracing/observability keys (the sub-request +// should stay tied to the caller's trace) and +// BifrostContextKeySkipPluginPipeline (whether the internal request runs the +// plugin pipeline is the caller's decision). +func ClearContextForInternalRequest(ctx *schemas.BifrostContext) { + // Key routing. + ctx.ClearValue(schemas.BifrostContextKeyGovernanceIncludeOnlyKeys) + ctx.ClearValue(schemas.BifrostContextKeyRoutingPinnedAPIKeyID) + ctx.ClearValue(schemas.BifrostContextKeyAPIKeyID) + ctx.ClearValue(schemas.BifrostContextKeyAPIKeyName) + ctx.ClearValue(schemas.BifrostContextKeyDirectKey) + ctx.ClearValue(schemas.BifrostContextKeySkipKeySelection) + // Body transport. + ctx.ClearValue(schemas.BifrostContextKeyUseRawRequestBody) + ctx.ClearValue(schemas.BifrostContextKeySendBackRawRequest) + ctx.ClearValue(schemas.BifrostContextKeySendBackRawResponse) + ctx.ClearValue(schemas.BifrostContextKeyPassthroughOverridesPresent) + ctx.ClearValue(schemas.BifrostContextKeyLargePayloadMode) + ctx.ClearValue(schemas.BifrostContextKeyLargeResponseMode) + ctx.ClearValue(schemas.BifrostContextKeyExtraHeaders) + ctx.ClearValue(schemas.BifrostContextKeyURLPath) +} + var supportedBaseProvidersSet = func() map[schemas.ModelProvider]struct{} { m := make(map[schemas.ModelProvider]struct{}, len(schemas.SupportedBaseProviders)) for _, p := range schemas.SupportedBaseProviders { @@ -415,6 +466,16 @@ func isPassthroughRequestType(reqType schemas.RequestType) bool { return reqType == schemas.PassthroughRequest || reqType == schemas.PassthroughStreamRequest } +// isResponsesLifecycleRequestType returns true for OpenAI Responses API lifecycle HTTP verbs. +func isResponsesLifecycleRequestType(reqType schemas.RequestType) bool { + switch reqType { + case schemas.ResponsesRetrieveRequest, schemas.ResponsesDeleteRequest, schemas.ResponsesCancelRequest, schemas.ResponsesInputItemsRequest: + return true + default: + return false + } +} + // IsFinalChunk returns true if the given context is a final chunk. func IsFinalChunk(ctx *schemas.BifrostContext) bool { if ctx == nil { diff --git a/core/version b/core/version index fdd3be6df54..266146b87cb 100644 --- a/core/version +++ b/core/version @@ -1 +1 @@ -1.6.2 +1.6.3 diff --git a/docs/benchmarking/run-your-own-benchmarks.mdx b/docs/benchmarking/run-your-own-benchmarks.mdx index b2a6365240a..c71b2ad073d 100644 --- a/docs/benchmarking/run-your-own-benchmarks.mdx +++ b/docs/benchmarking/run-your-own-benchmarks.mdx @@ -12,9 +12,13 @@ Want to see Bifrost's performance in your specific environment? The [**Bifrost B - **Custom Instance Sizes** - Test on your preferred AWS/GCP/Azure instances - **Your Workload Patterns** - Use your actual request/response sizes - **Different Configurations** - Compare various Bifrost settings -- **Provider Comparisons** - Benchmark against other AI gateways +- **Provider Comparisons** - Benchmark against other AI gateways or raw OpenAI - **Load Scenarios** - Test burst loads, sustained traffic, and endurance +The repo also ships two companion tools: +- **[mocker](https://github.com/maximhq/bifrost-benchmarking/tree/main/mocker)** — a mock LLM provider server with configurable latency, failures, and rate limits. Point your gateways at it to measure pure gateway overhead with zero API costs. +- **[hitter](https://github.com/maximhq/bifrost-benchmarking/tree/main/hitter)** — a load generator for stress-testing a single Bifrost deployment with realistic multi-model/streaming traffic. + > **💡 Open Source**: The benchmarking tool is completely open source! Feel free to submit pull requests if you think anything is missing or could be improved. --- @@ -23,9 +27,9 @@ Want to see Bifrost's performance in your specific environment? The [**Bifrost B Before running benchmarks, ensure you have: -- **Go 1.26.1+** installed on your testing machine +- **Go 1.24+** installed on your testing machine - **Bifrost instance** running and accessible -- **Target API providers** configured (OpenAI, Anthropic, etc.) +- **Target providers** configured in Bifrost (real providers, or the [mocker](https://github.com/maximhq/bifrost-benchmarking/tree/main/mocker) for cost-free runs) - **Network access** between benchmark tool and Bifrost - **Sufficient resources** on the testing machine to generate load @@ -48,39 +52,68 @@ go build benchmark.go This creates a `benchmark` executable (or `benchmark.exe` on Windows). -### **3. Run Your First Benchmark** +### **3. Configure Gateway Ports** + +Create a `.env` file in the repo root with the port of each gateway you plan to benchmark — the tool reads ports from here, not from flags: + +```env +BIFROST_PORT=8080 +OPENAI_API_KEY=sk-... # only needed when benchmarking raw OpenAI +``` + +To compare against other gateways, add their port variables too — the [repo README](https://github.com/maximhq/bifrost-benchmarking#readme) lists every supported gateway and its `.env` variable. + +### **4. Run Your First Benchmark** + +Either `-rate` (fixed RPS) or `-users` (fixed concurrency) is required: ```bash # Basic benchmark: 500 RPS for 10 seconds -./benchmark -provider bifrost -port 8080 +./benchmark -provider bifrost -rate 500 -# Custom benchmark: 1000 RPS for 30 seconds -./benchmark -provider bifrost -port 8080 -rate 1000 -duration 30 -output my_results.json +# Custom benchmark: 1000 RPS for 30 seconds +./benchmark -provider bifrost -rate 1000 -duration 30 -output my_results.json ``` +> **⚠️ Note**: Omitting `-provider` benchmarks **all** providers sequentially — including `openai`, which sends real requests to `api.openai.com` using your `OPENAI_API_KEY`. + --- ## Configuration Options -The benchmark tool offers extensive configuration through command-line flags: - ### **Basic Configuration** | Flag | Required | Description | Default | |------|----------|-------------|---------| -| `-provider ` | ✅ | Provider name (e.g., `bifrost`, `litellm`) | None | -| `-port ` | ✅ | Port number of your Bifrost instance | None | -| `-endpoint ` | ❌ | API endpoint path | `v1/chat/completions` | -| `-rate ` | ❌ | Requests per second | `500` | +| `-rate ` | ✅* | Requests per second (mutually exclusive with `-users`) | None | +| `-users ` | ✅* | Concurrent users to maintain (mutually exclusive with `-rate`) | None | +| `-provider ` | ❌ | Gateway to benchmark: `bifrost`, `openai`, or another supported gateway (full list in the [repo README](https://github.com/maximhq/bifrost-benchmarking#readme)); empty runs all | None (all) | | `-duration ` | ❌ | Test duration in seconds | `10` | | `-output ` | ❌ | Results output file | `results.json` | +| `-big-payload` | ❌ | Use a ~10KB request payload instead of the ~200B default | `false` | + +\* Exactly one of `-rate` or `-users` must be provided. ### **Advanced Configuration** | Flag | Description | Default | |------|-------------|---------| -| `-include-provider-in-request` | Include provider name in request payload | `false` | -| `-big-payload` | Use larger, more complex request payloads | `false` | +| `-timeout ` | Request timeout — set to duration + expected backend latency | `300` | +| `-cooldown ` | Cooldown between provider tests | `60` | +| `-model ` | Model to put in the request payload | `gpt-4o-mini` | +| `-host
` | Host address of the gateway servers | `localhost` | +| `-path ` | API path to hit (e.g. `chat/completions`, `embeddings`) | `chat/completions` | +| `-suffix ` | URL route suffix prepended to the path | `v1` | +| `-request-type ` | `chat` or `embedding` — controls payload shape | `chat` | +| `-prompt-file ` | File whose content is used as the prompt (for large-prompt tests) | `""` | +| `-ramp-up` | Gradually ramp users up (only with `-users`) | `false` | +| `-ramp-up-duration ` | Seconds to ramp from 1 to `-users` users | `0` | +| `-debug` | Detailed logging and periodic status updates | `false` | + +### **Rate vs. Users Mode** + +- **`-rate`** sends requests at a constant RPS regardless of response times — best for measuring throughput capacity and latency under a known load. +- **`-users`** keeps exactly N requests in flight at all times; as one completes, the next is dispatched. Throughput becomes ≈ `users / avg_latency` — best for simulating connection pools and realistic client behavior. --- @@ -91,7 +124,7 @@ The benchmark tool offers extensive configuration through command-line flags: Test standard performance with typical request sizes: ```bash -./benchmark -provider bifrost -port 8080 -rate 1000 -duration 60 -output basic_test.json +./benchmark -provider bifrost -rate 1000 -duration 60 -output basic_test.json ``` **Use Case**: General performance validation @@ -101,7 +134,7 @@ Test standard performance with typical request sizes: Push your instance to its limits: ```bash -./benchmark -provider bifrost -port 8080 -rate 5000 -duration 120 -output stress_test.json +./benchmark -provider bifrost -rate 5000 -duration 120 -output stress_test.json ``` **Use Case**: Capacity planning and SLA validation @@ -111,7 +144,7 @@ Push your instance to its limits: Test with bigger request/response sizes: ```bash -./benchmark -provider bifrost -port 8080 -rate 500 -duration 60 -big-payload=true -output large_payload.json +./benchmark -provider bifrost -rate 500 -duration 60 -big-payload -output large_payload.json ``` **Use Case**: Document processing, code generation workloads @@ -121,59 +154,64 @@ Test with bigger request/response sizes: Long-running stability test: ```bash -./benchmark -provider bifrost -port 8080 -rate 1000 -duration 1800 -output endurance_test.json +./benchmark -provider bifrost -rate 1000 -duration 1800 -timeout 2100 -output endurance_test.json ``` **Use Case**: Production readiness validation (30-minute test) -### **5. Comparative Benchmarking** +### **5. Concurrent Users with Ramp-Up** -Compare Bifrost against other providers: +Simulate realistic traffic that gradually builds: + +```bash +./benchmark -provider bifrost -users 500 -duration 600 -ramp-up -ramp-up-duration 120 -output rampup_test.json +``` + +**Use Case**: Realistic user behavior — ramps from 1 to 500 concurrent users over 2 minutes, then holds + +### **6. Comparative Benchmarking** + +Compare Bifrost against other gateways (each gateway's port comes from `.env`): ```bash # Test Bifrost -./benchmark -provider bifrost -port 8080 -rate 1000 -duration 60 -output bifrost_results.json +./benchmark -provider bifrost -rate 1000 -duration 60 -output bifrost_results.json -# Test LiteLLM -./benchmark -provider litellm -port 8000 -rate 1000 -duration 60 -output litellm_results.json +# Test another gateway (its port configured in .env — supported gateways listed in the repo README) +./benchmark -provider -rate 1000 -duration 60 -output gateway_results.json -# Test direct OpenAI (if available) -./benchmark -provider openai -port 443 -endpoint chat/completions -rate 1000 -duration 60 -output openai_results.json +# Test direct OpenAI (needs OPENAI_API_KEY in .env; note the explicit path) +./benchmark -provider openai -path v1/chat/completions -rate 100 -duration 60 -output openai_results.json ``` --- ## Understanding Results -The benchmark tool generates detailed JSON results with comprehensive metrics: +The benchmark tool writes per-provider metrics to the output file (keyed by provider, latest run per provider): ### **Key Metrics Explained** ```json { "bifrost": { - "request_counts": { - "total_sent": 30000, - "successful": 30000, - "failed": 0 - }, - "success_rate": 100.0, - "latency_metrics": { - "mean_ms": 245.5, - "p50_ms": 230.2, - "p99_ms": 520.8, - "max_ms": 845.3 - }, - "throughput_rps": 5000.0, - "memory_usage": { - "before_mb": 512.5, - "after_mb": 1312.8, - "peak_mb": 1405.2, - "average_mb": 1156.7 - }, + "requests": 30000, + "rate": 500.12, + "success_rate": 99.8, + "mean_latency_ms": 45.2, + "p50_latency_ms": 42.1, + "p99_latency_ms": 156.7, + "max_latency_ms": 203.4, + "throughput_rps": 498.5, "timestamp": "2025-01-14T10:30:00Z", - "status_codes": { - "200": 30000 + "status_code_counts": { + "200": 29940, + "500": 60 + }, + "server_peak_memory_mb": 256.7, + "server_avg_memory_mb": 189.3, + "drop_reasons": { + "HTTP 500": 60 } } } @@ -191,9 +229,10 @@ The benchmark tool generates detailed JSON results with comprehensive metrics: - **Mean**: Overall average performance **Memory Usage:** -- **Peak**: Maximum memory consumption -- **Average**: Sustained memory usage -- **After - Before**: Memory growth during test +- **Peak / Average**: server-side RSS sampled during the run — the tool finds the gateway process by its configured port, so run the benchmark on the same machine as the gateway to capture memory stats + +**Drop Reasons:** +- Categorized failure analysis (timeouts, HTTP errors, connection failures) --- @@ -237,39 +276,38 @@ Simulate traffic spikes: ```bash # Normal load -./benchmark -provider bifrost -port 8080 -rate 1000 -duration 300 -output normal_load.json +./benchmark -provider bifrost -rate 1000 -duration 300 -output normal_load.json # Burst load (simulate 5x spike) -./benchmark -provider bifrost -port 8080 -rate 5000 -duration 60 -output burst_load.json +./benchmark -provider bifrost -rate 5000 -duration 60 -output burst_load.json ``` ### **Multi-Instance Testing** -Test horizontal scaling: +Test horizontal scaling — environment variables override `.env`, so you can target multiple instances in parallel: ```bash # Instance 1 -./benchmark -provider bifrost-1 -port 8080 -rate 2500 -duration 120 -output instance_1.json & +BIFROST_PORT=8080 ./benchmark -provider bifrost -rate 2500 -duration 120 -output instance_1.json & -# Instance 2 -./benchmark -provider bifrost-2 -port 8081 -rate 2500 -duration 120 -output instance_2.json & +# Instance 2 +BIFROST_PORT=8081 ./benchmark -provider bifrost -rate 2500 -duration 120 -output instance_2.json & # Wait for both to complete wait ``` -### **Different Payload Sizes** +### **Embeddings Benchmarking** -Compare performance across payload sizes: +Benchmark embeddings endpoints, optionally with very large prompts from a file: ```bash -# Small payloads (default) -./benchmark -provider bifrost -port 8080 -rate 1000 -duration 60 -output small_payload.json - -# Large payloads -./benchmark -provider bifrost -port 8080 -rate 1000 -duration 60 -big-payload=true -output large_payload.json +./benchmark -provider bifrost -request-type embedding -path embeddings \ + -model text-embedding-3-small -prompt-file 10kbprompt.txt -rate 10 -duration 30 ``` +The repo root includes `10kbprompt.txt` and `50kbprompt.txt` as ready-made fixtures. + --- ## Continuous Benchmarking @@ -287,9 +325,9 @@ OUTPUT_DIR="benchmarks/$DATE" mkdir -p $OUTPUT_DIR # Run standard benchmarks -./benchmark -provider bifrost -port 8080 -rate 1000 -duration 300 -output "$OUTPUT_DIR/standard.json" -./benchmark -provider bifrost -port 8080 -rate 3000 -duration 180 -output "$OUTPUT_DIR/high_load.json" -./benchmark -provider bifrost -port 8080 -rate 500 -duration 600 -big-payload=true -output "$OUTPUT_DIR/large_payload.json" +./benchmark -provider bifrost -rate 1000 -duration 300 -output "$OUTPUT_DIR/standard.json" +./benchmark -provider bifrost -rate 3000 -duration 180 -output "$OUTPUT_DIR/high_load.json" +./benchmark -provider bifrost -rate 500 -duration 600 -big-payload -output "$OUTPUT_DIR/large_payload.json" echo "Benchmarks completed: $OUTPUT_DIR" ``` @@ -308,6 +346,9 @@ Monitor key metrics over time: ### **Common Issues** +**"Either --rate or --users flag must be provided":** +- Exactly one of `-rate` or `-users` is required; they are mutually exclusive. + **Connection Refused:** ```bash # Check if Bifrost is running @@ -316,7 +357,13 @@ curl http://localhost:8080/health # Verify port configuration netstat -an | grep 8080 ``` -- Check PORT is defined in `.env` file at root. +- Check the provider's port (e.g. `BIFROST_PORT`) is defined in the `.env` file at the repo root. + +**"No process found on port":** +- The gateway isn't running, or the `.env` port is wrong. The benchmark still runs; only memory stats are skipped. + +**"Attack for [Provider] timed out":** +- Raise `-timeout`; it must cover `duration + backend latency`. **High Error Rates:** - Check provider API key limits @@ -324,17 +371,12 @@ netstat -an | grep 8080 - Monitor upstream provider status - Reduce request rate for baseline test -**Memory Issues:** -- Monitor system resources during testing -- Check for memory leaks in long tests -- Adjust Bifrost pool sizes - **Inconsistent Results:** - Run multiple test iterations - Account for network variability - Use longer test durations (60+ seconds) - Isolate testing environment -- Try hitting gateway requests to a Mock provider +- Point the gateway at the repo's [mock provider](https://github.com/maximhq/bifrost-benchmarking/tree/main/mocker) to eliminate upstream variability --- diff --git a/docs/changelogs/helm-v2.1.26.mdx b/docs/changelogs/helm-v2.1.26.mdx new file mode 100644 index 00000000000..4cd138c9416 --- /dev/null +++ b/docs/changelogs/helm-v2.1.26.mdx @@ -0,0 +1,17 @@ +--- +title: "v2.1.26" +description: "Helm v2.1.26 changelog - 2026-07-06" +--- + + + +## Changelog + +- `bifrost.client.mcpServerAuthMode` (`headers` | `both` | `oauth`) and `bifrost.client.oauth2ServerConfig` (`issuerUrl`, `authCodeTtl`, `accessTokenTtl`, `disableVkIdentity`) to control how `/mcp` authenticates inbound MCP clients. `authCodeTtl` is capped at 900 seconds. Render into `client.mcp_server_auth_mode` and `client.oauth2_server_config`. +- ClickHouse logs store: set `storage.logsStore.type: clickhouse` with a `storage.logsStore.clickhouse` block (`host` required; optional `port`, `database`, `username`, `password`, `protocol`, `secure`, `dialTimeout`, `cluster`). +- `bedrock_mantle` provider with `bedrock_mantle_key_config` (`region` required; optional `access_key`, `secret_key`, `session_token`, `role_arn`, `external_id`, `session_name`). +- `deepseek` provider support via the generic provider passthrough. +- `toolExecutionTimeout` on `bifrost.mcp.clientConfigs[]` — a per-server override of the global `toolManagerConfig.toolExecutionTimeout`. Accepts a Go duration string (e.g. `"30s"`) or a bare integer treated as seconds. +- `expires_at` on `bifrost.governance.virtualKeys[]` — optional RFC3339 timestamp after which requests using the virtual key are rejected. Omit for a key that never expires. + + diff --git a/docs/cli-agents/claude-code.mdx b/docs/cli-agents/claude-code.mdx index 3b4d9967527..9e05c7704be 100644 --- a/docs/cli-agents/claude-code.mdx +++ b/docs/cli-agents/claude-code.mdx @@ -346,9 +346,18 @@ Run `/mcp` inside Claude Code. `bifrost` should appear as connected with a tool Unexpected identifier "Method". Raw body: Method Not Allowed ``` - This is cosmetic and has no functional impact. Claude Code probes `/register` (RFC 7591 Dynamic Client Registration) when you click **Re-authenticate**; Bifrost intentionally does not implement an OAuth stub for that probe, so the SDK logs the parse error. The `/mcp` connection itself works fine. + This appears whenever the [gateway auth mode](../mcp/gateway-auth) is `headers`: clicking **Re-authenticate** makes Claude Code probe `/register` (RFC 7591 Dynamic Client Registration), but registration and discovery aren't served in that mode, so the SDK logs the parse error. The `/mcp` connection itself keeps working. - To refresh tools from Bifrost, click **Reconnect** in the `/mcp` panel instead of Re-authenticate. See the upstream [Claude Code bug report](https://github.com/anthropics/claude-code/issues/46640) for context. + To make Re-authenticate work, switch `mcp_server_auth_mode` to `both` or `oauth` — Bifrost then serves real Dynamic Client Registration and the error disappears. If you're staying on `headers`, use **Reconnect** in the `/mcp` panel to refresh tools instead. See the upstream [Claude Code bug report](https://github.com/anthropics/claude-code/issues/46640) for context. + + + + Claude Code completed an OAuth flow, but the token it presented to `/mcp` was rejected. Two common causes: + + 1. **You recently switched `mcp_server_auth_mode`.** Claude Code caches OAuth state per server, and tokens issued under the old mode are no longer accepted. Remove and re-add the server (`claude mcp remove bifrost`, then add it again). + 2. **A VK header is being sent alongside the OAuth token.** Bifrost rejects requests carrying two credential types at once (`conflicting credentials`). This typically happens in `both` mode when the VK is configured under a non-standard header: Claude Code only treats a configured `Authorization` header as "use header auth" — with `x-bf-vk` or `X-Api-Key` it may still run the OAuth flow and then send the OAuth token *and* your VK header together, which Bifrost rejects. + + In `both` mode, configure the VK as `Authorization: Bearer ` (not `x-bf-vk` / `X-Api-Key`), or drop the header entirely and authenticate via OAuth. @@ -374,6 +383,10 @@ Run `/mcp` inside Claude Code. `bifrost` should appear as connected with a tool + + If Claude Code behaves unexpectedly after any change to Bifrost's MCP auth settings, remove and re-add the server — Claude Code caches auth tokens per MCP server, and stale cached credentials can survive Reconnect. + + ## Checklist 1. Ensure the model selected is same as you configured in the `settings.json`. diff --git a/docs/cli-agents/codex-cli.mdx b/docs/cli-agents/codex-cli.mdx index 5f08f099c0b..bee4335ead5 100644 --- a/docs/cli-agents/codex-cli.mdx +++ b/docs/cli-agents/codex-cli.mdx @@ -91,7 +91,7 @@ codex --model gemini/gemini-2.5-pro Bifrost supports the following providers with the `provider/model-name` format: -`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` +`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `deepseek`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` Non-OpenAI models **must support tool use** for Codex CLI to work properly. Codex CLI relies on tool calling for file operations, terminal commands, and code editing. Models without tool use support will fail on most operations. diff --git a/docs/cli-agents/cursor.mdx b/docs/cli-agents/cursor.mdx index d86e3f0d3d8..d050974b29d 100644 --- a/docs/cli-agents/cursor.mdx +++ b/docs/cli-agents/cursor.mdx @@ -75,7 +75,7 @@ mistral/mistral-large-latest Bifrost supports the following providers with the `provider/model-name` format: -`openai`, `anthropic`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` +`openai`, `anthropic`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `deepseek`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` Non-native models **must support tool use** for Cursor's agent mode and inline editing to work properly. Models without tool use support will only work for basic chat. diff --git a/docs/cli-agents/gemini-cli.mdx b/docs/cli-agents/gemini-cli.mdx index fd710b7f0bc..e2f3b945098 100644 --- a/docs/cli-agents/gemini-cli.mdx +++ b/docs/cli-agents/gemini-cli.mdx @@ -99,7 +99,7 @@ gemini -m groq/llama-3.3-70b-versatile Bifrost supports the following providers with the `provider/model-name` format: -`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` +`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `deepseek`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` Non-Google models **must support tool use** for Gemini CLI to work properly. Gemini CLI relies on tool calling for file operations, terminal commands, and code editing. Models without tool use support will fail on most operations. diff --git a/docs/cli-agents/librechat.mdx b/docs/cli-agents/librechat.mdx index 3e810e029ad..fc2993b8440 100644 --- a/docs/cli-agents/librechat.mdx +++ b/docs/cli-agents/librechat.mdx @@ -114,7 +114,7 @@ mistral/mistral-large-latest Bifrost supports the following providers with the `provider/model-name` format: -`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` +`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `deepseek`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` LibreChat connects to Bifrost via a single OpenAI-compatible endpoint. Bifrost handles routing to the correct provider based on the model name - no per-provider configuration needed in LibreChat. diff --git a/docs/cli-agents/open-webui.mdx b/docs/cli-agents/open-webui.mdx index 1db5bec4e03..f55bcc6426e 100644 --- a/docs/cli-agents/open-webui.mdx +++ b/docs/cli-agents/open-webui.mdx @@ -98,7 +98,7 @@ mistral/mistral-large-latest Bifrost supports the following providers with the `provider/model-name` format: -`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` +`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `deepseek`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` Open WebUI connects to Bifrost via a single OpenAI-compatible endpoint. Bifrost handles routing to the correct provider based on the model name - no per-provider configuration needed in Open WebUI. diff --git a/docs/cli-agents/opencode.mdx b/docs/cli-agents/opencode.mdx index dbd852e18c8..b6929853409 100644 --- a/docs/cli-agents/opencode.mdx +++ b/docs/cli-agents/opencode.mdx @@ -156,7 +156,7 @@ You can configure models from different providers with per-model options: Bifrost supports the following providers with the `provider/model-name` format: -`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` +`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `deepseek`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` Non-native models **must support tool use** for OpenCode to work properly. OpenCode relies on tool calling for file operations, terminal commands, and code editing. Models without tool use support will fail on most operations. diff --git a/docs/cli-agents/qwen-code.mdx b/docs/cli-agents/qwen-code.mdx index b2f85ccf356..0a228061283 100644 --- a/docs/cli-agents/qwen-code.mdx +++ b/docs/cli-agents/qwen-code.mdx @@ -110,7 +110,7 @@ Add multiple models to your `modelProviders.openai` array - they all use the sam Bifrost supports the following providers with the `provider/model-name` format: -`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` +`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `deepseek`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` Non-native models **must support tool use** for Qwen Code to work properly. Qwen Code relies on tool calling for file operations, terminal commands, and code editing. Models without tool use support will fail on most operations. diff --git a/docs/cli-agents/roo-code.mdx b/docs/cli-agents/roo-code.mdx index 4742acac955..7423bd4db42 100644 --- a/docs/cli-agents/roo-code.mdx +++ b/docs/cli-agents/roo-code.mdx @@ -70,7 +70,7 @@ mistral/mistral-large-latest Bifrost supports the following providers with the `provider/model-name` format: -`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` +`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `deepseek`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` Roo Code requires **native tool calling** (OpenAI-compatible function calling). Models without tool use support cannot be used with Roo Code. Ensure the model you select supports tool calling. diff --git a/docs/cli-agents/zed-editor.mdx b/docs/cli-agents/zed-editor.mdx index a76c97997ac..350f2e7181e 100644 --- a/docs/cli-agents/zed-editor.mdx +++ b/docs/cli-agents/zed-editor.mdx @@ -116,7 +116,7 @@ mistral/mistral-large-latest Bifrost supports the following providers with the `provider/model-name` format: -`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` +`openai`, `azure`, `gemini`, `vertex`, `bedrock`, `mistral`, `groq`, `cerebras`, `deepseek`, `cohere`, `perplexity`, `xai`, `ollama`, `openrouter`, `huggingface`, `nebius`, `parasail`, `replicate`, `vllm`, `sgl` Non-native models **must support tool use** for Zed's AI features (code actions, refactoring) to work properly. Models without tool use support will only work for basic chat and completions. diff --git a/docs/deployment-guides/config-json/providers.mdx b/docs/deployment-guides/config-json/providers.mdx index a52c9740839..83a57081ba8 100644 --- a/docs/deployment-guides/config-json/providers.mdx +++ b/docs/deployment-guides/config-json/providers.mdx @@ -419,6 +419,106 @@ When only `region` is set, Bifrost inherits credentials from the AWS SDK default + + +### AWS Bedrock Mantle + +Bedrock Mantle requires `bedrock_mantle_key_config` with a **required** `region` (no default). It authenticates with AWS SigV4 or an optional Bearer API key. See the [Bedrock Mantle provider page](../../providers/supported-providers/bedrock-mantle) for model-ID details. + + + + +```json +{ + "providers": { + "bedrock_mantle": { + "keys": [ + { + "name": "mantle-static", + "value": "", + "models": ["*"], + "weight": 1.0, + "bedrock_mantle_key_config": { + "region": "us-east-1", + "access_key": "env.AWS_ACCESS_KEY_ID", + "secret_key": "env.AWS_SECRET_ACCESS_KEY", + "session_token": "env.AWS_SESSION_TOKEN" + } + } + ] + } + } +} +``` + + + + +When only `region` is set, Bifrost inherits credentials from the AWS SDK default chain - IRSA, EC2 instance profile, or `AWS_*` env vars. + +```json +{ + "providers": { + "bedrock_mantle": { + "keys": [ + { + "name": "mantle-iam", + "value": "", + "models": ["*"], + "weight": 1.0, + "bedrock_mantle_key_config": { + "region": "us-east-1" + } + } + ] + } + } +} +``` + +To assume an IAM role before requests (works with both inherited and explicit credentials), add `role_arn` (and optionally `external_id` / `session_name`): + +```json +{ + "bedrock_mantle_key_config": { + "region": "us-west-2", + "role_arn": "env.AWS_ROLE_ARN", + "external_id": "env.AWS_EXTERNAL_ID", + "session_name": "bifrost-session" + } +} +``` + + + + +Set the top-level `value` to a Bedrock Mantle API key and leave the SigV4 credentials empty (`region` is still required). + +```json +{ + "providers": { + "bedrock_mantle": { + "keys": [ + { + "name": "mantle-api-key", + "value": "env.BEDROCK_MANTLE_API_KEY", + "models": ["*"], + "weight": 1.0, + "bedrock_mantle_key_config": { + "region": "us-east-1" + } + } + ] + } + } +} +``` + + + + + + ### Google Vertex AI @@ -534,6 +634,9 @@ These providers follow the same simple pattern - one or more keys with weights. "cerebras": { "keys": [{ "name": "cerebras-main", "value": "env.CEREBRAS_API_KEY", "models": ["*"], "weight": 1.0 }] }, + "deepseek": { + "keys": [{ "name": "deepseek-main", "value": "env.DEEPSEEK_API_KEY", "models": ["*"], "weight": 1.0 }] + }, "openrouter": { "keys": [{ "name": "openrouter-main", "value": "env.OPENROUTER_API_KEY", "models": ["*"], "weight": 1.0 }] }, diff --git a/docs/deployment-guides/config-json/schema-reference.mdx b/docs/deployment-guides/config-json/schema-reference.mdx index b6797057044..a3d38663e73 100644 --- a/docs/deployment-guides/config-json/schema-reference.mdx +++ b/docs/deployment-guides/config-json/schema-reference.mdx @@ -111,7 +111,7 @@ Full documentation: [Client Configuration](/deployment-guides/config-json/client Keyed by provider name. Each entry contains a `keys` array and optional `network_config`, `concurrency_and_buffer_size`, `proxy_config`. -Supported provider keys: `openai`, `anthropic`, `azure`, `bedrock`, `vertex`, `gemini`, `mistral`, `groq`, `cohere`, `perplexity`, `xai`, `cerebras`, `openrouter`, `nebius`, `fireworks`, `parasail`, `huggingface`, `replicate`, `ollama`, `vllm`, `sgl`, `elevenlabs`, `runway`. +Supported provider keys: `openai`, `anthropic`, `azure`, `bedrock`, `vertex`, `gemini`, `mistral`, `groq`, `cohere`, `perplexity`, `xai`, `cerebras`, `deepseek`, `openrouter`, `nebius`, `fireworks`, `parasail`, `huggingface`, `replicate`, `ollama`, `vllm`, `sgl`, `elevenlabs`, `runway`. Full documentation: [Provider Setup](/deployment-guides/config-json/providers). diff --git a/docs/deployment-guides/helm/providers.mdx b/docs/deployment-guides/helm/providers.mdx index 9c3ce426322..2c3af7dadd1 100644 --- a/docs/deployment-guides/helm/providers.mdx +++ b/docs/deployment-guides/helm/providers.mdx @@ -709,6 +709,11 @@ bifrost: - name: "cerebras-main" value: "env.CEREBRAS_API_KEY" weight: 1 + deepseek: + keys: + - name: "deepseek-main" + value: "env.DEEPSEEK_API_KEY" + weight: 1 openrouter: keys: - name: "openrouter-main" diff --git a/docs/deployment-guides/how-to/security-best-practices.mdx b/docs/deployment-guides/how-to/security-best-practices.mdx new file mode 100644 index 00000000000..9df2c0ff6dd --- /dev/null +++ b/docs/deployment-guides/how-to/security-best-practices.mdx @@ -0,0 +1,221 @@ +--- +title: "Security best practices" +description: "Best practices for hosting Bifrost on the public internet: strong dashboard credentials, enforced inference auth, locked-down CORS, and reverse-proxy security headers." +icon: "shield-halved" +--- + + +**Read this before you expose Bifrost to the internet.** Bifrost includes secure defaults, but a public deployment is only as safe as the controls you actually turn on. By default the dashboard and inference endpoints are reachable by anyone who can route to the host. The items below are the minimum hardening for any gateway that is reachable from outside your private network. + + +Most of these controls live on the **Security Settings** page in the dashboard at `/workspace/config/security`, or in your `config.json` (inference and CORS controls under the `client` block; dashboard credentials under `governance.auth_config`). The rest live in the reverse proxy in front of Bifrost. + +--- + +## 1. Use a strong dashboard password + +The dashboard can be protected with **Password protect the dashboard** on the Security Settings page (an admin username + password). Anyone who reaches the dashboard URL without credentials can read configuration, virtual keys, and logs, so this is the first thing to turn on for a public host. + + +**Password policy is enforced starting OSS v1.6.0 and Enterprise v1.5.0.** On these versions Bifrost validates the password both in the UI and on the server before saving, and rejects weak values with HTTP 400. On **earlier versions there was no strength check at all**. If you are running an older build, choose a strong password manually (and upgrade as soon as you can). + + +The enforced policy requires every dashboard password to have: + +- At least **12 characters** +- At least one **uppercase** letter +- At least one **lowercase** letter +- At least one **number** +- At least one **special character** + +```json +{ + "auth_config": { + "is_enabled": true, + "admin_username": "admin", + "admin_password": "env.BIFROST_ADMIN_PASSWORD" + } +} +``` + + +Reference the password from an environment variable or secret (`env.VAR_NAME`) instead of hardcoding a literal value in `config.json`. Env/secret references are stored as-is; literal passwords are hashed before storage. + + +For Enterprise deployments, prefer **SSO / OIDC** over a shared dashboard password so every operator who can change configuration is a known, traceable identity. See the [security hardening guide](/enterprise/moving-from-oss/security-hardening) and the SSO setup guides ([Okta](/enterprise/setting-up-okta), [Entra](/enterprise/setting-up-entra), [Keycloak](/enterprise/setting-up-keycloak), [Zitadel](/enterprise/setting-up-zitadel), [Google Workspace](/enterprise/setting-up-google-workspace)). + +--- + +## 2. Enforce authentication on inference + +By default, inference endpoints (`/v1/chat/completions`, `/v1/embeddings`, `/v1/images/generations`, and related endpoints) accept anonymous requests. On a public host that means anyone who finds the URL can spend against your provider keys. + +Turn on the **Enable Auth on Inference** toggle on the Security Settings page (labeled **Enforce Virtual Keys on Inference** in OSS). This requires every inference call to present a valid credential, such as a [Virtual Key](/features/governance/virtual-keys), API key, or user token, which Bifrost resolves to scoped upstream provider keys. Your raw provider keys never leave the gateway. + +```json +{ + "client": { + "enforce_auth_on_inference": true + } +} +``` + + +This is the main setting. The older fields `enforce_governance_header` and `enforce_scim_auth` are deprecated. Don't use them in new deployments. Changing this setting requires a Bifrost restart in Enterprise. + + +Once enforced, pair it with [budgets and rate limits](/features/governance/budget-and-limits) per virtual key so a runaway client can't burn through your provider spend even with valid credentials. + +--- + +## 3. Review the rest of the Security Settings page + +The Security Settings page (`/workspace/config/security`) exposes several more controls worth checking before going public: + +| Setting | Config key | Recommendation for public hosts | +|---|---|---| +| **Allow Direct API Keys** | `allow_direct_keys` | Keep **off** (default). When on, callers can pass their own provider key in a header (`x-bf-direct-key: true`), bypassing your registered key pool. | +| **Allowed Origins** | `allowed_origins` | Set an explicit list. Never `*` in production. A wildcard lets JavaScript from any page on the internet call your gateway. | +| **Allowed Headers** | `allowed_headers` | Narrow to the minimum your callers need (e.g. `Authorization`, `Content-Type`, your virtual-key and tracing headers). | +| **Required Headers** | `required_headers` | Optionally require headers on every request; missing ones are rejected with 400. | +| **Whitelisted Routes** | `whitelisted_routes` | Only add routes that must bypass auth. System routes (`/health`, login, etc.) are always whitelisted. | + +```json +{ + "client": { + "allow_direct_keys": false, + "allowed_origins": [ + "https://app.example.com", + "https://internal-dashboard.example.com" + ], + "allowed_headers": ["Authorization", "Content-Type", "X-Request-Id"] + } +} +``` + + +Changing `allowed_origins` or `allowed_headers` requires a Bifrost restart to take effect. + + +Enterprise deployments should also tighten the provider-forwarded `x-bf-eh-*` header allowlist (`header_filter_config`). See [Tighten both header allowlists](/enterprise/moving-from-oss/security-hardening) for details. + +--- + +## 4. Terminate TLS and serve from a reverse proxy + +Never expose Bifrost's HTTP port directly to the internet. Put a reverse proxy (NGINX, an Ingress controller, or a cloud load balancer) in front of it to terminate TLS, so all traffic, including dashboard logins, virtual keys, and prompts, is encrypted in transit. + +See the [Nginx reverse proxy guide](/deployment-guides/how-to/nginx-reverse-proxy) for streaming-safe proxy settings, and bind Bifrost itself to an internal interface so it is only reachable through the proxy. + +--- + +## 5. Send security headers from the reverse proxy + +Add hardening response headers at the proxy layer to defend the dashboard against clickjacking, MIME sniffing, and protocol downgrade. Bifrost is served behind the proxy, so this is the right place to set them once for every response. + + + + ```nginx + server { + listen 443 ssl; + server_name bifrost.example.com; + + # Force HTTPS for one year, including subdomains + add_header Strict-Transport-Security "max-age=31536000; includeSubDomains" always; + + # Clickjacking / iframe embedding protection + add_header X-Frame-Options "DENY" always; + add_header Content-Security-Policy "frame-ancestors 'none'" always; + + # Block MIME-type sniffing + add_header X-Content-Type-Options "nosniff" always; + + # Limit referrer leakage + add_header Referrer-Policy "strict-origin-when-cross-origin" always; + + location / { + proxy_pass http://bifrost_backend; + # ... streaming-safe proxy settings (see nginx guide) + } + } + ``` + + + + ```yaml + ingress: + enabled: true + className: nginx + annotations: + nginx.ingress.kubernetes.io/configuration-snippet: | + more_set_headers "Strict-Transport-Security: max-age=31536000; includeSubDomains"; + more_set_headers "X-Frame-Options: DENY"; + more_set_headers "Content-Security-Policy: frame-ancestors 'none'"; + more_set_headers "X-Content-Type-Options: nosniff"; + more_set_headers "Referrer-Policy: strict-origin-when-cross-origin"; + ``` + + + +| Header | Protects against | +|---|---| +| `Strict-Transport-Security` | Protocol downgrade / SSL-stripping attacks | +| `X-Frame-Options` / `Content-Security-Policy: frame-ancestors` | Clickjacking and embedding the dashboard in a hostile iframe | +| `X-Content-Type-Options: nosniff` | MIME-type sniffing | +| `Referrer-Policy` | Leaking dashboard URLs to third-party sites | + + +Use the `always` flag (NGINX) so headers are sent even on error responses. Only enable HSTS once you are confident HTTPS will stay on. Browsers cache it for the full `max-age`. + + +--- + +## 6. Restrict network exposure + +Network-level controls limit the impact of a misconfiguration: + +- **Don't publish the raw container port.** Expose only the reverse proxy; keep Bifrost on an internal network or `localhost` upstream. +- **Firewall / security groups.** Allow inbound traffic only on `443` (and `80` for the ACME/HTTP-to-HTTPS redirect). Block everything else. +- **Restrict the admin surface.** If only your team needs the dashboard, put it behind a VPN, an IP allowlist, or an identity-aware proxy rather than the open internet. +- **Run as non-root.** The official `maximhq/bifrost` image already runs as an unprivileged user. Keep it that way and avoid mounting host paths writable. + +--- + +## Hardening checklist + + + + 12+ chars with mixed case, number, and symbol. Upgrade to OSS v1.6.0 / Enterprise v1.5.0+ so the policy is enforced. + + + `enforce_auth_on_inference: true`: no anonymous path to a model. + + + `allow_direct_keys: false` unless you have a specific reason. + + + Explicit `allowed_origins`, never `*` in production. + + + No raw HTTP port exposed to the internet. + + + HSTS, frame-ancestors / X-Frame-Options, nosniff, Referrer-Policy. + + + Per-virtual-key limits before the first real request. + + + Firewall to 443, admin surface behind VPN/allowlist where possible. + + + +--- + +## Related guides + +- [Nginx reverse proxy](/deployment-guides/how-to/nginx-reverse-proxy) +- [Enterprise security hardening](/enterprise/moving-from-oss/security-hardening) +- [Virtual keys](/features/governance/virtual-keys) +- [Budgets and limits](/features/governance/budget-and-limits) +- [Security at Bifrost](/security): how Bifrost itself is built and scanned diff --git a/docs/docs.json b/docs/docs.json index 568250f1161..1b115a16c4d 100644 --- a/docs/docs.json +++ b/docs/docs.json @@ -133,9 +133,11 @@ "providers/supported-providers/anthropic", "providers/supported-providers/azure", "providers/supported-providers/bedrock", + "providers/supported-providers/bedrock-mantle", "providers/supported-providers/cerebras", "providers/supported-providers/cohere", "providers/supported-providers/databricks", + "providers/supported-providers/deepseek", "providers/supported-providers/elevenlabs", "providers/supported-providers/fireworks", "providers/supported-providers/gemini", @@ -192,6 +194,7 @@ "mcp/agent-mode", "mcp/code-mode", "mcp/gateway", + "mcp/gateway-auth", "mcp/tool-hosting", "mcp/filtering" ] @@ -620,6 +623,7 @@ "deployment-guides/how-to/install-make", "deployment-guides/how-to/multinode", "deployment-guides/how-to/nginx-reverse-proxy", + "deployment-guides/how-to/security-best-practices", "deployment-guides/how-to/airgapped", "deployment-guides/docker-tuning" ] @@ -970,6 +974,7 @@ "item": "Helm", "icon": "box", "pages": [ + "changelogs/helm-v2.1.26", "changelogs/helm-v2.1.25", "changelogs/helm-v2.1.24", "changelogs/helm-v2.1.23", @@ -1208,4 +1213,4 @@ "linkedin": "https://linkedin.com/company/maxim-ai" } } -} +} \ No newline at end of file diff --git a/docs/enterprise/adaptive-load-balancing.mdx b/docs/enterprise/adaptive-load-balancing.mdx index 89212251342..064673a8608 100644 --- a/docs/enterprise/adaptive-load-balancing.mdx +++ b/docs/enterprise/adaptive-load-balancing.mdx @@ -14,17 +14,19 @@ This page focuses on the technical implementation and performance characteristic ## Overview -**Adaptive Load Balancing** in Bifrost Enterprise automatically optimizes traffic distribution across providers and keys based on real-time performance metrics. The system operates at **two levels** - provider selection (direction) and key selection (route) - continuously monitoring error rates, latency, and throughput to dynamically adjust weights, ensuring optimal performance and reliability. + + Adaptive Load Balancing Dashboard + -### Key Features +**Adaptive Load Balancing** in Bifrost Enterprise automatically optimizes traffic distribution across providers and keys based on real-time performance metrics. The system operates at **two levels** - provider selection (direction) and key selection (route) - continuously monitoring error rates, latency, and throughput to dynamically adjust weights, ensuring optimal performance and reliability. | Feature | Description | |---------|-------------| | **Dynamic Weight Adjustment** | Automatically adjusts key weights based on performance metrics | | **Real-time Performance Monitoring** | Tracks error rates, latency, and success rates per model-key combination | -| **Cross-Node Synchronization** | Gossip protocol ensures consistent weight information across all cluster nodes | +| **Cross-Node Coordination** | Nodes share rate-limit (TPM) signals so an overloaded key is backed off fleet-wide within a region | | **Circuit Breaker Integration** | Temporarily removes poorly performing keys from rotation | -| **Fast Recovery** | Momentum-based scoring helps routes recover quickly after transient failures | +| **Fast Recovery** | Recovering routes are favored so they climb back quickly after transient failures | **Zero-overhead design**: All route selection logic adds less than **10 microseconds** to hot path latency. Weight calculations happen asynchronously every 5 seconds, so request routing uses pre-computed weights with minimal overhead. @@ -80,24 +82,21 @@ graph TB ## How Weight Calculation Works -Every 5 seconds, the system recalculates weights for all routes based on four factors: - -| Factor | Weight | Purpose | -|--------|--------|---------| -| **Error Penalty** | 50% | Penalizes routes with high error rates | -| **Latency Score** | 20% | Penalizes routes with abnormally slow responses | -| **Utilization Score** | 5% | Prevents overloading high-performing routes | -| **Momentum Bias** | Additive | Rewards routes that are recovering well | +Every 5 seconds, the system recalculates a weight for each route from its recent +performance. Three signals drive the score, in priority order: -The system combines these into a single score, then converts it to a weight between 1 and 1000. Lower penalties mean higher weights, which means more traffic. +| Factor | Role | Purpose | +|--------|------|---------| +| **Error Penalty** | Primary | Penalizes routes with high error rates | +| **Latency Score** | Secondary | Penalizes routes that are slow relative to their peers and to their own baseline | +| **Utilization** | Tuning | Discourages overloading any single high-performing route | -$$ -Score = (P_{error} \times 0.5) + (P_{latency} \times 0.2) + (P_{util} \times 0.05) - M_{momentum} -$$ - -$$ -Weight = W_{min} + (1 - Score) \times (W_{max} - W_{min}) -$$ +Which signals apply depends on the route's health: healthy routes are scored mainly +on errors and latency, while routes that are actively recovering are scored on latency +and recovery progress so they aren't held back by stale error history. The combined +score maps to a weight on a fixed scale - lower penalties mean higher weight, which +means more traffic - with a floor so no route is ever fully starved while it has a +chance to recover. ```mermaid flowchart LR @@ -105,27 +104,23 @@ flowchart LR E["Error Rate"] L["Latency"] U["Utilization"] - M["Momentum"] end - subgraph Scoring["Score Computation"] - EP["Error Penalty
50% weight"] - LP["Latency Score
20% weight"] - US["Utilization Score
5% weight"] - MS["Momentum Bias"] + subgraph Scoring["Health-aware Scoring"] + EP["Error Penalty
primary"] + LP["Latency Score
peer + baseline"] + US["Utilization
balancing"] end subgraph Output["Final Weight"] - NS["Normalized Score"] - FW["Route Weight
1 - 1000"] + NS["Combined Score"] + FW["Route Weight"] end E --> EP L --> LP U --> US - M --> MS EP & LP & US --> NS - MS --> NS NS --> FW ``` @@ -140,31 +135,103 @@ flowchart LR 3. **Real-time Dashboard**: Provides visibility into weight distribution, performance metrics (error rates, latency), state transitions, and actual vs expected traffic per route. - Adaptive Load Balancing Dashboard + Adaptive Load Balancing Dashboard -4. **Multi-Factor Scoring**: Routes are scored using 4 components - Error Penalty (50% weight, time-decayed), Latency Score (token-aware via MV-TACOS algorithm), Utilization Score (fair-share balancing), and Momentum (accelerates recovery after failures). +4. **Multi-Factor Scoring**: Routes are scored from error rate (the primary, time-decayed signal), a token-aware latency score (comparing a route both to its peers and to its own recent baseline), and fair-share utilization. Recovering routes are scored to favor quick, safe recovery. -5. **Smart Key Selection**: Uses weighted random selection with jitter (5% band) and 25% exploration probability to probe potentially recovered routes, rather than always picking the best route. +5. **Smart Key Selection**: Traffic is distributed probabilistically - higher-weight keys get proportionally more requests, but lower-weight keys keep a small share so potentially-recovered routes are continually re-probed instead of always picking the single best route. -6. **Performance Thresholds**: Specific triggers drive state transitions -> 2% error rate triggers Degraded, >5% error rate or TPM hit triggers Failed, <2% error with 50%+ expected traffic triggers Healthy. +6. **Performance Thresholds**: Pre-tuned error-rate and latency triggers drive state transitions - a route is marked Degraded at the first signs of trouble, Failed on sustained errors or a rate-limit hit, and promoted back to Healthy only after it has proven itself on live traffic. -The system is designed to be self-healing: it penalizes failing routes quickly, but also enables fast recovery (90% penalty reduction in 30 seconds) once issues are fixed. +The system is designed to be self-healing: it penalizes failing routes quickly, but also decays those penalties rapidly once issues are fixed, so a recovered route returns to full traffic within seconds. --- +## Configuration + +Adaptive load balancing ships **pre-tuned** - the scoring weights, thresholds, and recovery timings are not user-configurable by design. The operator controls are four switches: + +| Setting | Default | Effect | +|---------|---------|--------| +| **Provider selection** (`direction_selection_enabled`) | On | Whether the system picks the provider (Level 1). Off ⇒ the request's provider is honored as-is. | +| **Key selection** (`route_selection_enabled`) | On | Whether per-key (Level 2) selection is adaptive. Off ⇒ keys are chosen by static weighted-random (metrics are still tracked). | +| **Re-route failed providers** (`reroute_failed_directions`) | Off | If a pinned provider's direction is circuit-broken, re-route the request to a healthy provider for the same model. | +| **Prune failed fallbacks** (`prune_failed_fallbacks`) | Off | Drop circuit-broken providers from a request's configured fallback list. | + +The two selection switches default **on**; the two failed-direction behaviors are **opt-in**. All four take effect live (no restart) and propagate across the cluster. + +All four switches can be changed from the dashboard, via the API, or in `config.json`. + + + + +The load balancer settings page exposes the four switches; changes apply immediately across the cluster. + + + Load Balancer Settings + + + + + +Read and update the settings through `/api/load-balancer-config`: + +```bash +# Read the current settings +curl http://localhost:8080/api/load-balancer-config + +# Update the settings - always send all four fields +curl -X PUT http://localhost:8080/api/load-balancer-config \ + -H "Content-Type: application/json" \ + -d '{ + "direction_selection_enabled": true, + "route_selection_enabled": true, + "reroute_failed_directions": true, + "prune_failed_fallbacks": false + }' +``` + +The update is persisted, applied to the running nodes immediately, and broadcast to cluster peers. + + +The `PUT` body is a full replacement, not a patch: any field omitted from the request body is set to `false`. Always send all four fields - sending only the field you want to change silently turns off the others (including the default-on selection switches). + + + + + +Add a top-level `load_balancer_config` block: + +```json +{ + "load_balancer_config": { + "direction_selection_enabled": true, + "route_selection_enabled": true, + "reroute_failed_directions": true, + "prune_failed_fallbacks": false + } +} +``` + +Unlike the API, this block is presence-aware: omitted fields keep their current values, so you only need to list the switches you want to change. Settings saved through the dashboard or the API take precedence over `config.json` values. + + + + +## Scope & Limitations + +- **Per-node weights**: Each node load-balances on its own observed metrics. The only signal shared across nodes is a rate-limit (TPM) backoff, and only within the same region - there is no global weight consensus or cross-region coordination. This is deliberate: latency and error profiles differ per region, so importing another region's metrics would pollute a node's view of route health. +- **~5-second adaptation**: Weight and state changes lag live traffic by up to one recompute cycle. Immediate per-request resilience (key rotation and fallback failover) is handled separately and is not subject to this delay. +- **Optimistic cold start**: A brand-new key or provider enters at full weight and competes at roughly fair share before it has been measured, then self-corrects within a cycle or two. +- **Relative, not absolute**: Routes are ranked against their peers, not against a fixed latency or cost target. The system is not cost-, org-, or session-aware, and does not accept manual per-key weights for the adaptive path - those concerns are handled by [governance routing](/providers/provider-routing). + +--- + ## Next Steps - - - Contact your Bifrost Enterprise representative to enable adaptive load balancing for your deployment - - - Use the dashboard to observe how weights adapt to real traffic patterns - - - Review route state transitions and weight adjustments to understand system behavior - - +- **[Provider Routing](/providers/provider-routing)** - How adaptive load balancing composes with governance rules and the Model Catalog +- **[Circuit Breaker](./circuit-breaker)** - Header-signal-driven failover to a backup provider when a primary endpoint degrades +- **[Clustering](./clustering)** - Multi-node deployments and the gossip layer behind cross-node load balancer signals diff --git a/docs/features/governance/complexity-router.mdx b/docs/features/governance/complexity-router.mdx index a55f0fbe3f0..8590dc7535d 100644 --- a/docs/features/governance/complexity-router.mdx +++ b/docs/features/governance/complexity-router.mdx @@ -6,14 +6,14 @@ icon: "sliders" ## Overview -The Complexity Router analyzes each incoming request and assigns it one of four tiers - **Simple**, **Medium**, **Complex**, or **Reasoning** - based on the content of the latest user message, conversation history, and system prompt. The result is exposed as a flat string variable (`complexity_tier`) in Bifrost's CEL routing engine, so you can write routing rules like: +The Complexity Router analyzes incoming requests and assigns a tier - **Simple**, **Medium**, **Complex**, or **Reasoning** - when the latest user message contains a clear complexity signal. The result is exposed as a flat string variable (`complexity_tier`) in Bifrost's CEL routing engine, so you can write routing rules like: ```cel complexity_tier == "REASONING" complexity_tier in ["COMPLEX", "REASONING"] ``` -This lets you route simple greetings to a fast, cheap model and deep reasoning tasks to a frontier model — automatically, with no changes to your application code. The algorithm is fast and deterministic: it runs entirely in-process using pre-compiled keyword matching, adds less than 1 ms to request latency, and makes zero external calls. +This lets you route simple greetings to a fast, cheap model and deep reasoning tasks to a frontier model — automatically, with no changes to your application code. When the latest user message does not match a configured signal, Bifrost leaves `complexity_tier` unknown and keeps the request on its existing routing path instead of guessing. The algorithm is fast and deterministic: it runs entirely in-process using pre-compiled keyword matching, adds less than 1 ms to request latency, and makes zero external calls. ![Complexity Router Configuration](../../media/architecture-complexity-router.png) @@ -23,41 +23,33 @@ This lets you route simple greetings to a fast, cheap model and deep reasoning t ### Scoring dimensions -Every request produces a score between 0.0 and 1.0. The analyzer starts with a weighted score across five dimensions detected by scanning the last user message: +Classified requests produce a score between 0.0 and 1.0. The analyzer starts with a weighted score across five dimensions detected by scanning the last user message: | Dimension | Weight | What it measures | |---|---|---| | Code presence | 30% | Code, debugging, and programming artifacts | | Reasoning markers | 25% | Analytical and multi-step reasoning language | | Technical terms | 25% | Architecture, infra, and operational terminology | -| Token count | 10% | Prompt length (longer → higher score) | -| Simple indicators | −5% | Greetings, trivial queries (dampener, subtracted) | +| Token count | 10% | Prompt length, used only after a signal or continuation is detected | +| Simple indicators | -5% | Greetings, definitions, translation requests, and other straightforward asks | -The simple indicators dimension subtracts from the score - it acts as a dampener, not a floor. This means a short, conversational prompt like "hi, how are you?" can reach a near-zero score even if it technically contains other weak signals. The dampener is reduced to near-zero when the prompt is long (≥30 words) or contains two or more other strong signals, so it does not suppress genuinely complex requests. +Simple indicators are a small downward nudge, not an override. A simple-only request can classify as **Simple** with a `0.00` score, while a request that also has strong code, technical, or reasoning signals can still rise into a higher tier. ### System prompt contribution -The system prompt is scanned for code, technical, and simple signals, and its contribution is weighted at **25% of the user-message signal** for those three dimensions. This provides soft lexical context — for example, a system prompt describing a coding assistant nudges code scores up — but it never drives the token count, reasoning markers, or tier override. +When the latest user message already has a code, technical, or reasoning signal, the system prompt is scanned for code and technical signals. Its contribution is weighted at **25% of the user-message signal** for those dimensions. This provides soft lexical context — for example, a system prompt describing a coding assistant nudges code scores up — but it never creates a classification by itself. ### Conversation context blending -For multi-turn conversations, the score blends the current message with history from up to the last 10 user turns (recency-weighted: earlier turns count less): +For multi-turn conversations, prior context is used only when the latest user message has its own code, technical, or reasoning signal, or when the latest message is an explicit continuation phrase such as "do it", "retry", "continue", or "go ahead". In those cases, the score blends the current message with history from up to the last 10 user turns (recency-weighted: earlier turns count less): - **Default blend:** 60% last message + 40% conversation history -- **Referential follow-up blend:** 35% last message + 65% conversation history +- **Continuation blend:** 35% last message + 65% conversation history -A message is treated as a referential follow-up when it is short (≤6 words), contains phrases like "do it", "retry", "continue", or "go ahead", and the conversation history has a meaningful complexity score. In that case, the follow-up inherits most of its score from prior context rather than being classified as Simple on its own. +An explicit continuation phrase inherits most of its score from prior context rather than being classified on its own. A low-signal latest message that is not an explicit continuation does not inherit old context; `complexity_tier` remains unknown and routing falls through to the original model. The final score is `max(last_message_score, weighted_blend)` — the current message always sets a floor. -### Output complexity floor - -Some requests are hard not because the reasoning is especially deep, but because the output being asked for is broad or exhaustive. Prompts like "list every AWS service and explain each one with examples" can receive a built-in score floor even when the normal weighted score is only moderate. - -The analyzer looks for internal markers such as exhaustive enumeration ("list every", "all possible"), comprehensiveness cues ("comprehensive", "in detail"), and elaboration asks ("explain each", "with examples"). Limiting qualifiers like "briefly", "top 5", or "keep it short" reduce this boost. - -This output-complexity floor is built in — it is not currently exposed as a user-configurable keyword list. - ### Reasoning override When two or more **reasoning keywords** are detected in the last user message, the tier is forced to **Reasoning** regardless of the numeric score. The same override applies when one strong reasoning keyword appears alongside strong code or technical signals. @@ -175,7 +167,7 @@ curl -X POST http://localhost:8080/api/governance/complexity-analyzer-config/res | `keywords.code_keywords` | string[] | Yes | built-in defaults | Signals for code/debugging/programming requests | | `keywords.reasoning_keywords` | string[] | Yes | built-in defaults | Strong reasoning triggers — matches can force the Reasoning tier | | `keywords.technical_keywords` | string[] | Yes | built-in defaults | Architecture/infra/operations signals | -| `keywords.simple_keywords` | string[] | Yes | built-in defaults | Phrases that dampen the complexity score | +| `keywords.simple_keywords` | string[] | Yes | built-in defaults | Phrases that slightly reduce complexity and classify straightforward requests | Each keyword list requires at least one entry. Keywords are normalized to lowercase and deduplicated on save. Changes are hot-reloaded with no restart required. @@ -193,7 +185,7 @@ When `governance.complexity_analyzer_config` is present in `config.json`, the de Each list controls a different part of the scoring signal. Understanding what they do helps you tune routing for your domain. -The default keyword lists are tuned for common request patterns and are a good starting point for most deployments. For domain-specific traffic, add or remove keywords based on the prompts your users actually send so the tiers match your routing strategy. +The default keyword lists are tuned for common request patterns and are a good starting point for most deployments. For domain-specific traffic, add or remove keywords based on the prompts your users actually send so the tiers match your routing strategy. Bifrost stores and displays the exact keywords you configure; internally, normal word-based keywords and phrases also match common word forms, such as `debug`, `debugging`, and `debugged`. Punctuation-heavy terms such as `ci/cd` still use literal matching. @@ -211,7 +203,7 @@ Type a keyword or phrase and press **Enter** to add it. Click the × on any tag | **Code keywords** | Each match contributes to the Code dimension (30% weight) | Add domain-specific tooling, frameworks, or file types your users frequently mention | | **Reasoning keywords** | Strong triggers — two or more matches, or one match alongside strong code/technical signals, forces the Reasoning tier regardless of score | Narrow this list to the phrases that truly demand your most capable model. The default list is intentionally conservative. | | **Technical keywords** | Each match contributes to the Technical dimension (25% weight) | Add industry-specific jargon (e.g. "SOC 2", "HIPAA", "proration") relevant to your product | -| **Simple keywords** | Each match subtracts from the score (dampener) | Add domain-specific phrases that signal trivial intent in your app context | +| **Simple keywords** | Each match slightly reduces the score and can classify straightforward requests as Simple | Add domain-specific phrases that signal trivial intent in your app context | The **reasoning keywords** list gates the tier-override path, not just scoring. Adding broad terms like "explain" or "analyze" will push many requests to Reasoning. Prefer specific multi-word phrases like "step by step" or "root cause analysis". @@ -351,11 +343,11 @@ Test complexity routing with a single team before enabling it globally: ## Observability -Complexity analysis is recorded in the routing log for every request where analysis ran. In the log detail view, look at the **Routing Decision Logs** section backed by `routing_engine_logs`. +Complexity analysis is recorded in the routing log for every request where analysis ran and produced a tier. In the log detail view, look at the **Routing Decision Logs** section backed by `routing_engine_logs`. ![Routing Logs with Complexity Tier](../../media/ui-routing-logs-complexity.png) -You will see log lines like `Complexity: tier=REASONING score=0.38 words=25`. Logs use the emitted uppercase tier string, while the UI displays the same tier as normal-case text. This lets you audit how traffic is being distributed and spot mis-classifications to tune thresholds or keyword lists. +You will see log lines like `Complexity: tier=REASONING score=0.38 words=25`. If no configured complexity signal matches the latest user message, the routing log records that complexity analysis was skipped and routing continues on the existing path. This lets you audit how traffic is being distributed and spot mis-classifications to tune thresholds or keyword lists. --- @@ -363,9 +355,9 @@ You will see log lines like `Complexity: tier=REASONING score=0.38 words=25`. Lo ### Rule not matching when complexity_tier is set -If the routing rule uses `complexity_tier` and the request is not matching, make sure the request contains analyzable user text. A system prompt by itself is not enough — the analyzer needs a text-bearing user prompt to classify. +If the routing rule uses `complexity_tier` and the request is not matching, make sure the latest user message contains analyzable user text and at least one configured signal. A system prompt by itself is not enough — the analyzer needs a text-bearing user prompt to classify. -If analysis is unavailable (for example the body could not be parsed, or the user content is not text-only), `complexity_tier` is treated as **unknown** by the CEL evaluator. The rule does not match and evaluation falls through to the next rule. This is intentional: complexity rules silently degrade rather than blocking requests. +If analysis is unavailable (for example the body could not be parsed, the user content is not text-only, or no configured complexity signal matches the latest user message), the complexity-dependent rule does not match and evaluation falls through to the next rule. This is intentional: complexity rules silently degrade rather than blocking requests. ### Which request types are supported diff --git a/docs/features/governance/mcp-tools.mdx b/docs/features/governance/mcp-tools.mdx index 015baf69b41..618541decb2 100644 --- a/docs/features/governance/mcp-tools.mdx +++ b/docs/features/governance/mcp-tools.mdx @@ -29,6 +29,10 @@ For each MCP client associated with a Virtual Key, you can specify the allowed t - **Leave tool list empty**: All tools from that client will be **blocked**. - **Do not configure a client**: All tools from that client will be **blocked** (if other clients are configured). + +Inactive or [expired](./virtual-keys#key-expiry) Virtual Keys are rejected at MCP tool execution time with a `403`, regardless of their tool configuration. + + ## Setting MCP Tool Restrictions diff --git a/docs/features/governance/virtual-keys.mdx b/docs/features/governance/virtual-keys.mdx index 37f15362bd9..8ba63adc01d 100644 --- a/docs/features/governance/virtual-keys.mdx +++ b/docs/features/governance/virtual-keys.mdx @@ -48,6 +48,8 @@ Virtual Keys are the primary governance entity in Bifrost. Users and application - **Team**: Assign to existing team (mutually exclusive with customer) - **Customer**: Assign to existing customer (mutually exclusive with team) +**Expiry** (optional): Pick **Never**, a preset (30 min to 7 days), or a custom date and time. See [Key Expiry](#key-expiry). + 3. Click **Create Virtual Key**
@@ -534,6 +536,104 @@ curl -X PUT http://localhost:8080/api/config \ When the governance header is enforced, the request will be rejected if the `x-bf-vk` header is not present. +### Key Expiry + +Virtual keys can optionally carry an expiry timestamp. Once the expiry passes, requests using the key are rejected with a `403` and the reason `Virtual key has expired` — the key is not deleted or deactivated, so it stays visible for auditing and can be restored at any time. + +- **No expiry by default** — keys without `expires_at` never expire. +- **Fail closed** — both LLM inference and MCP tool execution are blocked once the key expires. +- **Inactive wins** — a key that is both inactive and expired is rejected as inactive. +- **Restore anytime** — extend the expiry to a future timestamp or clear it; access resumes immediately. + + + + +1. Go to **Virtual Keys** and create or edit a key +2. In the **Expiry** section, pick **Never**, a preset (**30 min**, **1 hour**, **24 hours**, **7 days**), or choose a custom date and time from the calendar + +![Virtual Key Expiry Picker](../../media/ui-virtual-key-expiry.png) + +Expired keys show an **Expired** badge in the virtual keys table. + + + + +**Create with expiry:** +```bash +curl -X POST http://localhost:8080/api/governance/virtual-keys \ + -H "Content-Type: application/json" \ + -d '{ + "name": "Contractor API Key", + "provider_configs": [ + { + "provider": "openai", + "allowed_models": ["gpt-4o-mini"], + "key_ids": ["*"] + } + ], + "expires_at": "2026-08-01T00:00:00Z" + }' +``` + +**Set or extend expiry on an existing key:** +```bash +curl -X PUT http://localhost:8080/api/governance/virtual-keys/{vk_id} \ + -H "Content-Type: application/json" \ + -d '{"expires_at": "2026-09-01T00:00:00Z"}' +``` + +**Clear expiry (key never expires again):** +```bash +curl -X PUT http://localhost:8080/api/governance/virtual-keys/{vk_id} \ + -H "Content-Type: application/json" \ + -d '{"expires_at": ""}' +``` + +Timestamps must be RFC3339 and in the future; otherwise the API returns `400`. On update, omitting `expires_at` leaves the current expiry unchanged. + +**Expired key rejection:** +```json +{ + "type": "virtual_key_blocked", + "status_code": 403, + "error": { + "message": "Virtual key has expired" + } +} +``` + + + + +```json +{ + "governance": { + "virtual_keys": [ + { + "id": "vk-contractor", + "name": "Contractor API Key", + "value": "sk-bf-*", + "provider_configs": [ + { + "provider": "openai", + "allowed_models": ["gpt-4o-mini"], + "key_ids": ["*"] + } + ], + "expires_at": "2026-08-01T00:00:00Z" + } + ] + } +} +``` + + +For config-managed virtual keys the file is the source of truth: removing `expires_at` from the file clears the expiry on the next sync. + + + + + ### Authentication and Virtual Keys Virtual keys and HTTP authentication are **independent layers** that can work together: diff --git a/docs/mcp/auth/oauth.mdx b/docs/mcp/auth/oauth.mdx index 6d1361ca307..da543867ab7 100644 --- a/docs/mcp/auth/oauth.mdx +++ b/docs/mcp/auth/oauth.mdx @@ -226,10 +226,16 @@ Status values: - `pending` — admin hasn't authorized yet - `authorized` — token is valid and active - `failed` — authorization failed or token is invalid +- `revoked` — the token was revoked (via DELETE); the config row is retained with no live token ### Automatic refresh -Bifrost refreshes the access token automatically before expiration using the stored refresh token. No action required. +Bifrost refreshes access tokens automatically using the stored refresh token, in two layers: + +- **In the background** — a worker periodically refreshes tokens that are about to expire, so active clients always have a valid token ready. +- **On use** — if a token is already expired when a request needs it, Bifrost refreshes it inline before forwarding the request. + +Background refresh only runs while the MCP client is enabled. Disabling a client pauses it; on re-enable, the token is refreshed on first use. If a client stays disabled long enough for the provider to expire the idle refresh token, re-authorization is required. ### Rotation @@ -241,7 +247,7 @@ The MCP client edit flow lets you rotate `client_id` / `client_secret` while pre curl -X DELETE http://localhost:8080/api/oauth/config/oauth_cfg_abc123 ``` -This revokes the token with the OAuth provider (if a revocation endpoint is configured), deletes the token from Bifrost, and removes the OAuth configuration. +This deletes the stored token from Bifrost and marks the OAuth configuration `revoked` (the config row is kept, not deleted). Bifrost does **not** call the upstream provider's revocation endpoint — revoke at the provider's dashboard if you need the upstream token invalidated there. --- @@ -334,6 +340,7 @@ See [Reverse Proxy configuration →](../../deployment-guides/config-json/client - Check that the refresh token is still valid (some providers expire refresh tokens after long idle) +- If the client was disabled for a long stretch, background refresh was paused for it — the refresh token may have expired at the provider in the meantime - Verify scopes are still sufficient - Re-authorize: `DELETE /api/oauth/config/{id}` then create a new client diff --git a/docs/mcp/auth/overview.mdx b/docs/mcp/auth/overview.mdx index 3157f73afdb..7322a3cfef0 100644 --- a/docs/mcp/auth/overview.mdx +++ b/docs/mcp/auth/overview.mdx @@ -82,7 +82,7 @@ Per-user auth keys every credential against an **identity**. The mode is derived | Mode | How it's set | Notes | | --------- | ------------------------------------------------------------------------------------------------------- | -------------------------------------------------- | | `user` | Bifrost's auth middleware populates `BifrostContextKeyUserID` (signed-in user via SSO), **or** the caller sends a VK that is owned by a user — Bifrost auto-promotes the VK's owner onto the context | Enterprise SSO and enterprise user-owned VKs only | -| `vk` | Caller sends `x-bf-vk` (or `Authorization: Bearer …` / `x-api-key`) and the VK resolves but is **not** owned by a user | Typical non-enterprise pattern, or enterprise VKs that aren't tied to a person | +| `vk` | Caller sends `x-bf-vk` (or `Authorization: Bearer …` / `x-api-key` / `x-goog-api-key`) and the VK resolves but is **not** owned by a user | Typical non-enterprise pattern, or enterprise VKs that aren't tied to a person | | `session` | Caller sends `x-bf-mcp-session-id: ` and re-sends the same value on later calls | Useful when there is no VK and no SSO | Priority: `user` > `vk` > `session`. If multiple are present (e.g., a user-owned VK both resolves a VK ID **and** promotes a user ID), Bifrost picks the highest-priority and ignores the rest for credential lookup. This means a user-owned VK always lands in `user` mode — the credential and any auth flow are bound to the user, not the VK. diff --git a/docs/mcp/auth/per-user-headers.mdx b/docs/mcp/auth/per-user-headers.mdx index be249516b51..497ad3575b5 100644 --- a/docs/mcp/auth/per-user-headers.mdx +++ b/docs/mcp/auth/per-user-headers.mdx @@ -258,7 +258,7 @@ The admin sample values you supplied (`user_headers`) didn't pass upstream auth. -The VK isn't resolving. Confirm the VK exists and the caller is sending it under one of `x-bf-vk`, `Authorization: Bearer …`, or `x-api-key`. If you're behind a proxy that strips `Authorization`, switch the caller to `x-bf-vk`. +The VK isn't resolving. Confirm the VK exists and the caller is sending it under one of `x-bf-vk`, `Authorization: Bearer …`, `x-api-key`, or `x-goog-api-key`. If you're behind a proxy that strips `Authorization`, switch the caller to `x-bf-vk`. diff --git a/docs/mcp/auth/per-user-oauth.mdx b/docs/mcp/auth/per-user-oauth.mdx index e85b8b45b7c..482fc20662e 100644 --- a/docs/mcp/auth/per-user-oauth.mdx +++ b/docs/mcp/auth/per-user-oauth.mdx @@ -14,7 +14,9 @@ If a single shared admin token is fine, use [OAuth 2.0](./oauth) instead. This auth type is only valid for **HTTP** and **SSE** connections. -Bifrost is **not** an OAuth 2.1 Authorization Server. The MCP Gateway (`/mcp`) does not advertise its own `.well-known` endpoints or run a consent screen for inbound MCP clients. Identity is asserted by the caller via headers (or upstream SSO), and auth happens **lazily** — on the first tool call that needs an upstream token. +This page covers **upstream** per-user OAuth — Bifrost holding a token *for* an upstream MCP service on behalf of each end-user, resolved **lazily** on the first tool call that needs it. Identity is asserted by the caller via headers (or upstream SSO). + +This is separate from Bifrost acting as an OAuth 2.1 Authorization Server *for inbound* `/mcp` clients (browser consent + `.well-known` discovery), which is covered in [Gateway Authentication](../gateway-auth). | | Server-level OAuth (`oauth`) | Per-user OAuth (`per_user_oauth`) | diff --git a/docs/mcp/gateway-auth.mdx b/docs/mcp/gateway-auth.mdx new file mode 100644 index 00000000000..26e05aa30cb --- /dev/null +++ b/docs/mcp/gateway-auth.mdx @@ -0,0 +1,252 @@ +--- +title: "Gateway Authentication" +description: "How MCP clients authenticate to Bifrost's /mcp endpoint — virtual key headers or browser-based OAuth 2.1." +icon: "key" +--- + +## Overview + +When Bifrost acts as an [MCP Gateway](./gateway), external MCP clients connect to its `/mcp` endpoint. This page covers how those **inbound clients authenticate to Bifrost**. + +There are two ways a client can present itself: + +- **Header credentials** — a virtual key, API key, or session token sent as a request header. Simple to script, ideal for backend and machine-to-machine use. +- **OAuth 2.1** — Bifrost acts as an OAuth authorization server, and the client connects through a browser consent flow, receiving a short-lived JWT. Ideal for interactive clients like Claude Desktop, Claude Code, or Cursor, where pasting a raw key into client config is awkward. + + +This page is about authenticating clients **to** Bifrost. For how Bifrost authenticates **to upstream MCP servers** it connects to, see the outbound [Authentication](./auth/overview) guides instead — that's the opposite direction. + + +--- + +## Authentication Modes + +A single setting, `mcp_server_auth_mode`, controls which credential types `/mcp` accepts: + +| Mode | Header credentials (VK / api-key / session) | Bifrost-issued JWT | Discovery endpoints | +|---|---|---|---| +| `headers` (default) | Accepted | — | Disabled | +| `both` | Accepted | Accepted | Enabled | +| `oauth` | Rejected | Accepted | Enabled | + +- **`headers`** — `/mcp` accepts header credentials only. The OAuth surface and `.well-known` discovery endpoints are not served. +- **`both`** — `/mcp` accepts header credentials **and** Bifrost-issued JWTs. Discovery is served so OAuth clients can connect. +- **`oauth`** — `/mcp` accepts Bifrost-issued JWTs only; header credentials are rejected. + +A request must carry **exactly one** credential type. If an OAuth access token and a header credential (`x-bf-vk`, `X-Api-Key`, or a Bearer VK) arrive on the same request, Bifrost rejects it with a `conflicting credentials` error — even in `both` mode. `both` means either credential is accepted, not both at once. + + +In `oauth` mode, clients that authenticate with a virtual key or API key header can no longer reach `/mcp`. Use `both` if you need header and OAuth clients to coexist. + + +--- + +## How the OAuth Connect Flow Works + +When `mcp_server_auth_mode` is `both` or `oauth`, Bifrost is a full OAuth 2.1 authorization server for the `/mcp` resource. A client that doesn't yet have a token discovers the server, registers itself, and walks the user through a browser consent step. + +```mermaid +sequenceDiagram + participant C as MCP Client + participant U as User (Browser) + participant B as Bifrost + + C->>B: GET /mcp (no token) + B-->>C: 401 + WWW-Authenticate (resource metadata URL) + C->>B: GET /.well-known/oauth-protected-resource + authorization-server + C->>B: POST /oauth2/register (Dynamic Client Registration) + C->>U: Open /oauth2/authorize (PKCE) in browser + U->>B: Consent page — choose identity + B-->>U: Redirect with authorization code + U-->>C: Code delivered to client redirect URI + C->>B: POST /oauth2/token (code + PKCE verifier) + B-->>C: access_token (JWT) + refresh_token + C->>B: GET/POST /mcp (Authorization: Bearer ) + B-->>C: Tools available +``` + +The flow follows current OAuth standards so off-the-shelf MCP clients work without custom code: + +- **Protected resource metadata** (RFC 9728) — the `401` response points clients at the discovery documents. +- **Dynamic Client Registration** (RFC 7591) — clients self-register; no manual client setup. +- **PKCE** (S256) — public clients authenticate without a shared secret. +- **Resource indicators** (RFC 8707) — tokens are bound to the `/mcp` resource via the `aud` claim. + +--- + +## Identity Modes at Consent + +During the consent step, Bifrost shows the user how they can identify themselves. The page header names the connecting client — for example **"Claude Code wants to connect"** — and offers the modes available for your deployment: + +![MCP OAuth consent page with identity options](/media/ui-oauth-consent.png) + +| Mode | Binds the grant to | Availability | +|---|---|---| +| **Virtual key** | A virtual key you paste on the consent page | Always available | +| **Session** | A server-minted, anonymous session identity | Only when `enforce_auth_on_inference` is `false` | +| **User** | Your signed-in dashboard user | Requires SSO / SCIM | + +- **Virtual key** carries the same governance (budgets, rate limits, tool scoping) the key already has — the JWT simply represents that key. +- **Session** is an anonymous identity for development and open deployments. It is unavailable once `enforce_auth_on_inference` is on. +- **User** binds the grant to the authenticated person, so per-user upstream tool authorizations unify under one identity. + + +**User mode requires SSO/SCIM** (enterprise). When no identity provider is configured, the consent page offers only virtual key and session modes. + + +--- + +## Configuration + + + + +1. Open **Config** and go to the **MCP** settings. +2. Set **MCP Server Auth Mode** to `headers`, `both`, or `oauth`. +3. When using `both` or `oauth`, optionally set the **OAuth Server** settings: an **Issuer URL** (required for multi-host deployments), and the **Authorization Code** and **Access Token** lifetimes. +4. Click **Save**. + +![MCP server auth mode and OAuth server settings](/media/ui-mcp-server-auth-mode.png) + + + + +```bash +curl -X PUT http://localhost:8080/api/config \ + -H "Content-Type: application/json" \ + -d '{ + "client_config": { + "mcp_server_auth_mode": "both", + "oauth2_server_config": { + "issuer_url": "https://bifrost.example.com", + "auth_code_ttl": 300, + "access_token_ttl": 600 + } + } + }' +``` + + + + +```json +{ + "client": { + "mcp_server_auth_mode": "both", + "oauth2_server_config": { + "issuer_url": "https://bifrost.example.com", + "auth_code_ttl": 300, + "access_token_ttl": 600 + } + } +} +``` + +| Field | Type | Required | Description | +|-------|------|----------|-------------| +| `mcp_server_auth_mode` | string | No | `headers` (default), `both`, or `oauth`. | +| `oauth2_server_config.issuer_url` | string | No | Stable public URL advertised as the issuer in discovery docs and the JWT `iss` claim. Required for multi-host deployments; single-host can omit it (falls back to the request `Host`). Supports `env.MY_VAR` syntax. | +| `oauth2_server_config.auth_code_ttl` | integer | No | Authorization code lifetime in seconds (default `300`, max `900` = 15 minutes). | +| `oauth2_server_config.access_token_ttl` | integer | No | Issued JWT lifetime in seconds (default `600`). | + + + + + +`oauth2_server_config` only applies when `mcp_server_auth_mode` is `both` or `oauth`. The RSA signing key used for JWTs is generated automatically the first time it's needed — no setup required. It is persisted to the database, so it survives restarts and is shared across all replicas; previously issued tokens stay valid after a restart. + + +--- + +## Connecting a Client + +With `both` or `oauth` enabled, point the MCP client at Bifrost's `/mcp` URL — for example `https://bifrost.example.com/mcp`. No key needs to be pasted into the client config. + +1. The client hits `/mcp`, gets a `401`, and discovers the authorization server. +2. It registers itself and opens a browser to the consent page. +3. The user chooses an identity (virtual key, session, or user). +4. The client receives a token and connects; aggregated tools become available. + +When the access token expires, the client uses its refresh token to obtain a new one silently — the browser step happens only once. + +--- + +## Managing Grants + +Each completed OAuth connection is a **grant** — a refresh-token lineage Bifrost issued to a client. The **OAuth Grants** page lists them: the client, the bound identity, when the grant was created, and when it was last used, with a **Revoke** action. + +![OAuth Grants table with revoke action](/media/ui-oauth-grants.png) + +Revoking a grant stops its refresh token from rotating immediately, so the client can no longer renew access. + + +A revoked grant's current access token is a short-lived JWT that keeps working on `/mcp` until it expires (up to `access_token_ttl`, default `600` seconds). After that the client is fully cut off and must reconnect through the consent flow. + +This is deliberate: a holder of the virtual key or user credentials can always start a new authorized session, so invalidating the access token mid-flight adds little security while costing a per-request lookup. Lower `access_token_ttl` for a tighter window. + + + +The **OAuth Grants** page lists credentials Bifrost **issued to clients** for inbound `/mcp` access. This is distinct from [MCP Sessions](./sessions), which tracks per-user credentials Bifrost holds for **upstream** MCP servers. + + +--- + +## Token Lifetime & Revocation + +- **Access tokens** are JWTs valid for `access_token_ttl` (default 600s). On every `/mcp` request Bifrost validates the JWT — signature, expiry, issuer, and audience — and confirms the bound identity (the virtual key or user) still exists and is active. The identity check reads an in-memory cache, so in the common case it adds no database round-trip (vk-mode falls back to a single store lookup only on a cache miss). +- **Refresh tokens** rotate on each use and have no fixed expiry. A token is invalidated by rotation, by the bound virtual key or user becoming inactive or deleted, or by an explicit revoke. +- An explicit revoke marks the grant's refresh-token row revoked, so renewals stop immediately — but it does **not** remove the bound identity, so the identity check still passes and the already-issued access token keeps working until it expires. Lower `access_token_ttl` to shorten that window. +- **Deleting** the bound virtual key or user is stronger than revoking a grant: it revokes the grant immediately *and* the identity check rejects the grant's already-issued access token on its next `/mcp` request, instead of letting it live out its TTL. +- Enabling `disable_vk_identity` (require identity-provider login) cuts off all **virtual-key**–mode grants immediately — they are rejected at `/mcp` and denied on refresh — so those clients must re-authenticate as a user. Only applies in `oauth` mode with an identity provider configured. +- Enabling `enforce_auth_on_inference` blocks session-mode (anonymous) tokens at `/mcp`, but does **not** delete or invalidate their grants — the grants remain in the database and become valid again if enforcement is later disabled. To remove a session-mode grant permanently, revoke it on the **OAuth Grants** page. + +--- + +## Discovery Endpoints + +When discovery is enabled (`both` or `oauth`), Bifrost serves the standard documents MCP clients fetch automatically: + +| Endpoint | Purpose | +|---|---| +| `GET /.well-known/oauth-protected-resource` | Protected resource metadata (RFC 9728) | +| `GET /.well-known/oauth-authorization-server` | Authorization server metadata (RFC 8414) | +| `GET /.well-known/jwks.json` | Public signing keys for JWT verification (RFC 7517) | + +In `headers` mode these endpoints return `404`. + +--- + +## Troubleshooting + +### Discovery returns 404 +**Symptom:** A client can't discover the authorization server; `.well-known` endpoints return `404`. +**Cause:** `mcp_server_auth_mode` is `headers`, so the OAuth surface is disabled. +**Fix:** Set the mode to `both` or `oauth`. + +### Header credential rejected on /mcp +**Symptom:** A virtual key or API key that worked before now gets a `401`. +**Cause:** `mcp_server_auth_mode` is `oauth`, which accepts JWTs only. +**Fix:** Use `both` to accept header credentials alongside OAuth. + +### Conflicting credentials on /mcp +**Symptom:** `conflicting credentials: an OAuth token and a virtual key header were both provided`. +**Cause:** The client completed an OAuth flow but is also configured with a VK header, so both arrive on one request. Claude Code does this in `both` mode when the VK is set under `x-bf-vk` / `X-Api-Key` instead of `Authorization`. +**Fix:** Send only one credential — remove the VK header, or configure it as `Authorization: Bearer ` so the client uses header auth and skips OAuth. + +### Client keeps re-opening the browser +**Symptom:** The consent flow runs on every connection. +**Cause:** The client isn't persisting its tokens, or its refresh token was revoked. +**Fix:** Confirm the client stores its credentials; check the **OAuth Grants** page to see whether the grant was revoked. + +### Token rejected after issuer change +**Symptom:** Previously issued tokens fail validation after changing `issuer_url`. +**Cause:** The `iss` claim in existing tokens no longer matches the configured issuer. +**Fix:** Clients reconnect to obtain tokens with the new issuer; set a stable `issuer_url` up front in multi-host deployments. + +--- + +## Next Steps + +- **[Bifrost as an MCP Gateway](./gateway)** — Expose aggregated tools to external MCP clients. +- **[MCP Sessions](./sessions)** — Inspect and manage per-user credentials for upstream servers. +- **[Virtual Keys](../features/governance/virtual-keys)** — Govern budgets, rate limits, and tool scope for the identities behind grants. diff --git a/docs/mcp/gateway.mdx b/docs/mcp/gateway.mdx index d54c97b2e1d..51e335c0716 100644 --- a/docs/mcp/gateway.mdx +++ b/docs/mcp/gateway.mdx @@ -120,6 +120,10 @@ Include any required Virtual Key authentication headers if governance is enabled Bifrost supports per-Virtual Key MCP servers, allowing you to expose different tools to different clients. + +Header credentials are one of two ways clients authenticate to `/mcp`. Clients can also connect through a browser-based OAuth flow — see [Gateway Authentication](./gateway-auth) for the `mcp_server_auth_mode` setting and the OAuth connect flow. + + ### Global Server (No Virtual Key) When `enforce_auth_on_inference` is `false`, requests without a Virtual Key use the global MCP server with all available tools. @@ -131,6 +135,12 @@ When using Virtual Keys, each VK gets its own MCP server with filtered tools bas **Authenticate with Virtual Key:** ```bash +# Via x-bf-vk header +curl -X POST http://localhost:8080/mcp \ + -H "x-bf-vk: vk_your_virtual_key" \ + -H "Content-Type: application/json" \ + -d '{"jsonrpc": "2.0", "id": 1, "method": "tools/list"}' + # Via Authorization header curl -X POST http://localhost:8080/mcp \ -H "Authorization: Bearer vk_your_virtual_key" \ @@ -143,9 +153,9 @@ curl -X POST http://localhost:8080/mcp \ -H "Content-Type: application/json" \ -d '{"jsonrpc": "2.0", "id": 1, "method": "tools/list"}' -# Via x-bf-vk header +# Via x-goog-api-key header curl -X POST http://localhost:8080/mcp \ - -H "x-bf-vk: vk_your_virtual_key" \ + -H "x-goog-api-key: vk_your_virtual_key" \ -H "Content-Type: application/json" \ -d '{"jsonrpc": "2.0", "id": 1, "method": "tools/list"}' ``` @@ -307,9 +317,11 @@ The MCP Server dynamically updates its tool registry from the tool manager. ## Per-User Auth on the Gateway -When at least one upstream MCP server is configured with `per_user_oauth` or `per_user_headers`, the `/mcp` endpoint serves per-user credentials lazily. Bifrost is **not** an OAuth Authorization Server — there is no `.well-known/oauth-protected-resource` discovery and no inbound consent screen. Inbound MCP clients identify themselves via headers: +When at least one upstream MCP server is configured with `per_user_oauth` or `per_user_headers`, the `/mcp` endpoint serves per-user credentials lazily, keyed to the inbound caller's identity. How that identity is established depends on the [gateway auth mode](./gateway-auth): in the default `headers` mode, clients identify themselves via headers (below); in `both` / `oauth` mode, a Bifrost-issued JWT carries the identity instead. + +In `headers` mode, inbound MCP clients identify themselves via headers: -- `x-bf-vk: ` (or `Authorization: Bearer ` / `x-api-key: `) — VK-mode identity +- `x-bf-vk: ` (or `Authorization: Bearer ` / `x-api-key: ` / `x-goog-api-key: `) — VK-mode identity - `x-bf-mcp-session-id: ` — session-mode identity (client-asserted, must be re-sent on every call) - Enterprise SSO — user-mode identity, attached automatically by the auth middleware @@ -328,7 +340,7 @@ Who can open and complete that URL depends on the flow's identity mode (frozen w See [Flow mode and access rules →](./auth/overview#flow-mode-and-access-rules) for the full per-mode behavior. -Claude Code may proactively POST `/register` (RFC 7591 DCR) on `claude mcp add` and log `SDK auth failed: …`. Bifrost intentionally does not implement an OAuth stub for that probe; the `/mcp` connection itself works. See the [Claude Code bug report](https://github.com/anthropics/claude-code/issues/46640) for context. +In the default `headers` [auth mode](./gateway-auth), Claude Code may proactively POST `/oauth2/register` (RFC 7591 DCR) on `claude mcp add` and log `SDK auth failed: …` — discovery and registration aren't served in that mode, so the probe has no endpoint to hit. The `/mcp` connection itself still works. In `both` / `oauth` mode the probe is handled by Bifrost's authorization server and the message doesn't appear. See the [Claude Code bug report](https://github.com/anthropics/claude-code/issues/46640) for context. See [Per-User OAuth →](./auth/per-user-oauth) and [Per-User Headers →](./auth/per-user-headers) for the full flows and identity options, and [MCP Sessions →](./sessions) for managing the resulting credentials. diff --git a/docs/media/provider-dashboard-deepseek.png b/docs/media/provider-dashboard-deepseek.png new file mode 100644 index 00000000000..8facad60d72 Binary files /dev/null and b/docs/media/provider-dashboard-deepseek.png differ diff --git a/docs/media/ui-load-balancing-dashboard.png b/docs/media/ui-load-balancing-dashboard.png new file mode 100644 index 00000000000..c8a8baed987 Binary files /dev/null and b/docs/media/ui-load-balancing-dashboard.png differ diff --git a/docs/media/ui-load-balancing-metrics.png b/docs/media/ui-load-balancing-metrics.png new file mode 100644 index 00000000000..f3df17a3f4b Binary files /dev/null and b/docs/media/ui-load-balancing-metrics.png differ diff --git a/docs/media/ui-load-balancing-settings.png b/docs/media/ui-load-balancing-settings.png new file mode 100644 index 00000000000..1ba641b618d Binary files /dev/null and b/docs/media/ui-load-balancing-settings.png differ diff --git a/docs/media/ui-load-balancing.png b/docs/media/ui-load-balancing.png deleted file mode 100644 index 217234ee292..00000000000 Binary files a/docs/media/ui-load-balancing.png and /dev/null differ diff --git a/docs/media/ui-mcp-server-auth-mode.png b/docs/media/ui-mcp-server-auth-mode.png new file mode 100644 index 00000000000..d6b96d98393 Binary files /dev/null and b/docs/media/ui-mcp-server-auth-mode.png differ diff --git a/docs/media/ui-oauth-consent.png b/docs/media/ui-oauth-consent.png new file mode 100644 index 00000000000..d8db9f14365 Binary files /dev/null and b/docs/media/ui-oauth-consent.png differ diff --git a/docs/media/ui-oauth-grants.png b/docs/media/ui-oauth-grants.png new file mode 100644 index 00000000000..d2b5188a524 Binary files /dev/null and b/docs/media/ui-oauth-grants.png differ diff --git a/docs/media/ui-virtual-key-expiry.png b/docs/media/ui-virtual-key-expiry.png new file mode 100644 index 00000000000..dfd5a68d31a Binary files /dev/null and b/docs/media/ui-virtual-key-expiry.png differ diff --git a/docs/openapi/openapi.json b/docs/openapi/openapi.json index c93a106396d..3771feaa534 100644 --- a/docs/openapi/openapi.json +++ b/docs/openapi/openapi.json @@ -38070,10 +38070,107 @@ "get": { "operationId": "getMCPClients", "summary": "List MCP clients", - "description": "Returns a list of all configured MCP clients with their tools and connection state.", + "description": "Returns a paginated list of configured MCP clients with their tools and connection state.\nSupports case-insensitive name search and exact-match filtering by connection type, auth type,\ncode-mode, and enabled/disabled status. Multi-value filters accept a comma-separated list and\nuse OR semantics within a field.\n", "tags": [ "MCP" ], + "parameters": [ + { + "name": "limit", + "in": "query", + "description": "Maximum number of clients to return (1–100, default 25).", + "schema": { + "type": "integer", + "minimum": 1, + "maximum": 100, + "default": 25 + } + }, + { + "name": "offset", + "in": "query", + "description": "Number of clients to skip.", + "schema": { + "type": "integer", + "minimum": 0 + } + }, + { + "name": "search", + "in": "query", + "description": "Case-insensitive search by client name.", + "schema": { + "type": "string" + } + }, + { + "name": "server", + "in": "query", + "description": "Filter to a single client by its exact client_id.", + "schema": { + "type": "string" + } + }, + { + "name": "connection_type", + "in": "query", + "description": "Comma-separated connection types to include (OR semantics).", + "schema": { + "type": "string", + "example": "http,sse" + } + }, + { + "name": "auth_type", + "in": "query", + "description": "Comma-separated auth types to include (OR semantics).", + "schema": { + "type": "string", + "example": "oauth,per_user_oauth" + } + }, + { + "name": "state", + "in": "query", + "description": "Comma-separated runtime connection states to include (OR semantics),\nresolved against live engine state. `connected` matches clients the engine\ncurrently reports as connected; `disconnected` matches everything else.\n", + "schema": { + "type": "string", + "example": "connected" + } + }, + { + "name": "all_virtual_keys", + "in": "query", + "description": "When true, include clients that are open to all virtual keys (allow_on_all_virtual_keys). ORs with virtual_keys.", + "schema": { + "type": "boolean" + } + }, + { + "name": "virtual_keys", + "in": "query", + "description": "Comma-separated virtual key IDs; includes clients explicitly assigned to any of them. ORs with all_virtual_keys.", + "schema": { + "type": "string" + } + }, + { + "name": "code_mode", + "in": "query", + "description": "Filter by code-mode clients. Omit for no filter.", + "schema": { + "type": "boolean" + } + }, + { + "name": "disabled", + "in": "query", + "description": "Filter by disabled status — true returns disabled clients, false returns enabled clients. Omit for no filter.", + "schema": { + "type": "boolean" + } + } + ], "security": [ { "ManagementBearerAuth": [] @@ -38085,14 +38182,54 @@ "content": { "application/json": { "schema": { - "type": "array", - "items": { - "$ref": "#/components/schemas/MCPClient" + "type": "object", + "description": "Paginated list of MCP clients.", + "required": [ + "clients", + "count", + "total_count", + "limit", + "offset" + ], + "properties": { + "clients": { + "type": "array", + "items": { + "$ref": "#/components/schemas/MCPClient" + } + }, + "count": { + "type": "integer", + "description": "Number of clients returned in this page" + }, + "total_count": { + "type": "integer", + "format": "int64", + "description": "Total number of clients matching the query (before pagination)" + }, + "limit": { + "type": "integer", + "description": "Page size used for the response" + }, + "offset": { + "type": "integer", + "description": "Page offset used for the response" + } } } } } }, + "400": { + "description": "Bad request", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/BifrostError" + } + } + } + }, "500": { "description": "Internal server error", "content": { @@ -61045,6 +61182,7 @@ "perplexity", "replicate", "cerebras", + "deepseek", "gemini", "openrouter", "elevenlabs", @@ -61906,7 +62044,7 @@ "prompt_cache_retention": { "type": "string", "enum": [ - "in-memory", + "in_memory", "24h" ], "description": "Prompt cache retention policy" @@ -71494,7 +71632,7 @@ }, "description": "Headers to capture in log metadata. Values are extracted from incoming requests and stored in the metadata field of log entries. Case-insensitive matching. No restart required." }, - "mcp_external_base_url": { + "mcp_external_client_url": { "oneOf": [ { "type": "string", @@ -71516,7 +71654,66 @@ "additionalProperties": false } ], - "description": "Public base URL for OAuth callbacks and discovery metadata when Bifrost runs behind a reverse proxy. Overrides the host derived from the incoming request. Supports env var syntax (\"env.MY_VAR\"). Configurable via UI, API, or config.json.\n" + "description": "Public base URL Bifrost uses as the redirect_uri when acting as an OAuth client to upstream MCP servers (Notion, Jira, etc.). Set when Bifrost's callback endpoint is reached via a different URL than its server-side metadata. Supports env var syntax (\"env.MY_VAR\").\n" + }, + "mcp_server_auth_mode": { + "type": "string", + "enum": [ + "headers", + "both", + "oauth" + ], + "default": "headers", + "description": "How /mcp authenticates inbound MCP clients. 'headers' (default): VK/api-key/session headers only, discovery disabled. 'both': accepts header credentials and Bifrost-issued JWTs, discovery enabled. 'oauth': Bifrost JWTs only — disables VK/header MCP access.\n" + }, + "oauth2_server_config": { + "type": "object", + "description": "OAuth2 authorization server settings for /mcp. Only relevant when mcp_server_auth_mode is 'both' or 'oauth'.\n", + "properties": { + "issuer_url": { + "oneOf": [ + { + "type": "string", + "description": "Plain URL or env var reference (e.g. \"env.MY_VAR\")" + }, + { + "type": "object", + "properties": { + "value": { + "type": "string" + }, + "env_var": { + "type": "string" + }, + "from_env": { + "type": "boolean" + } + }, + "additionalProperties": false + } + ], + "description": "Stable public URL advertised as the OAuth2 AS issuer in discovery documents and the JWT iss claim. Required for multi-host deployments; single-host can omit it (falls back to the request Host header). Supports env var syntax (\"env.MY_VAR\").\n" + }, + "auth_code_ttl": { + "type": "integer", + "minimum": 1, + "maximum": 900, + "default": 300, + "description": "Lifetime of the single-use authorization code in seconds (default 300, max 900 = 15 minutes)." + }, + "access_token_ttl": { + "type": "integer", + "minimum": 1, + "default": 600, + "description": "Lifetime of the issued JWT Bearer token in seconds (default 600)." + }, + "disable_vk_identity": { + "type": "boolean", + "default": false, + "description": "Require identity-provider login: virtual-key identity is removed from the consent flow, and existing virtual-key-mode grants are rejected at /mcp and denied on refresh. Only meaningful when mcp_server_auth_mode is 'oauth' and an identity provider is configured. Anonymous session identity is governed separately by enforce_auth_on_inference.\n" + } + }, + "additionalProperties": false } } }, @@ -71761,7 +71958,7 @@ }, "description": "Headers to capture in log metadata. Values are extracted from incoming requests and stored in the metadata field of log entries. Case-insensitive matching. No restart required." }, - "mcp_external_base_url": { + "mcp_external_client_url": { "oneOf": [ { "type": "string", @@ -71783,7 +71980,66 @@ "additionalProperties": false } ], - "description": "Public base URL for OAuth callbacks and discovery metadata when Bifrost runs behind a reverse proxy. Overrides the host derived from the incoming request. Supports env var syntax (\"env.MY_VAR\"). Configurable via UI, API, or config.json.\n" + "description": "Public base URL Bifrost uses as the redirect_uri when acting as an OAuth client to upstream MCP servers (Notion, Jira, etc.). Set when Bifrost's callback endpoint is reached via a different URL than its server-side metadata. Supports env var syntax (\"env.MY_VAR\").\n" + }, + "mcp_server_auth_mode": { + "type": "string", + "enum": [ + "headers", + "both", + "oauth" + ], + "default": "headers", + "description": "How /mcp authenticates inbound MCP clients. 'headers' (default): VK/api-key/session headers only, discovery disabled. 'both': accepts header credentials and Bifrost-issued JWTs, discovery enabled. 'oauth': Bifrost JWTs only — disables VK/header MCP access.\n" + }, + "oauth2_server_config": { + "type": "object", + "description": "OAuth2 authorization server settings for /mcp. Only relevant when mcp_server_auth_mode is 'both' or 'oauth'.\n", + "properties": { + "issuer_url": { + "oneOf": [ + { + "type": "string", + "description": "Plain URL or env var reference (e.g. \"env.MY_VAR\")" + }, + { + "type": "object", + "properties": { + "value": { + "type": "string" + }, + "env_var": { + "type": "string" + }, + "from_env": { + "type": "boolean" + } + }, + "additionalProperties": false + } + ], + "description": "Stable public URL advertised as the OAuth2 AS issuer in discovery documents and the JWT iss claim. Required for multi-host deployments; single-host can omit it (falls back to the request Host header). Supports env var syntax (\"env.MY_VAR\").\n" + }, + "auth_code_ttl": { + "type": "integer", + "minimum": 1, + "maximum": 900, + "default": 300, + "description": "Lifetime of the single-use authorization code in seconds (default 300, max 900 = 15 minutes)." + }, + "access_token_ttl": { + "type": "integer", + "minimum": 1, + "default": 600, + "description": "Lifetime of the issued JWT Bearer token in seconds (default 600)." + }, + "disable_vk_identity": { + "type": "boolean", + "default": false, + "description": "Require identity-provider login: virtual-key identity is removed from the consent flow, and existing virtual-key-mode grants are rejected at /mcp and denied on refresh. Only meaningful when mcp_server_auth_mode is 'oauth' and an identity provider is configured. Anonymous session identity is governed separately by enforce_auth_on_inference.\n" + } + }, + "additionalProperties": false } } }, @@ -73876,6 +74132,14 @@ "is_active": { "type": "boolean" }, + "expires_at": { + "type": [ + "string", + "null" + ], + "format": "date-time", + "description": "Expiry timestamp. Requests using this virtual key are rejected once it passes. Null or absent means the key never expires." + }, "provider_configs": { "type": "array", "items": { @@ -74329,6 +74593,11 @@ "calendar_aligned": { "type": "boolean", "default": false + }, + "expires_at": { + "type": "string", + "format": "date-time", + "description": "Optional expiry timestamp. Must be in the future. Omit for a key that never expires." } } }, @@ -74436,6 +74705,10 @@ "reset_budget_usage": { "type": "boolean", "description": "When true, resets usage for virtual-key and provider-config budget records reconciled by this update." + }, + "expires_at": { + "type": "string", + "description": "RFC3339 timestamp to set a new expiry (must be in the future), empty string to clear an existing expiry, omitted to leave it unchanged." } } }, diff --git a/docs/openapi/paths/management/mcp.yaml b/docs/openapi/paths/management/mcp.yaml index ef483c19e5d..e6b86e480a6 100644 --- a/docs/openapi/paths/management/mcp.yaml +++ b/docs/openapi/paths/management/mcp.yaml @@ -76,9 +76,79 @@ clients: get: operationId: getMCPClients summary: List MCP clients - description: Returns a list of all configured MCP clients with their tools and connection state. + description: | + Returns a paginated list of configured MCP clients with their tools and connection state. + Supports case-insensitive name search and exact-match filtering by connection type, auth type, + code-mode, and enabled/disabled status. Multi-value filters accept a comma-separated list and + use OR semantics within a field. tags: - MCP + parameters: + - name: limit + in: query + description: Maximum number of clients to return (1–100, default 25). + schema: + type: integer + minimum: 1 + maximum: 100 + default: 25 + - name: offset + in: query + description: Number of clients to skip. + schema: + type: integer + minimum: 0 + - name: search + in: query + description: Case-insensitive search by client name. + schema: + type: string + - name: server + in: query + description: Filter to a single client by its exact client_id. + schema: + type: string + - name: connection_type + in: query + description: Comma-separated connection types to include (OR semantics). + schema: + type: string + example: http,sse + - name: auth_type + in: query + description: Comma-separated auth types to include (OR semantics). + schema: + type: string + example: oauth,per_user_oauth + - name: state + in: query + description: | + Comma-separated runtime connection states to include (OR semantics), + resolved against live engine state. `connected` matches clients the engine + currently reports as connected; `disconnected` matches everything else. + schema: + type: string + example: connected + - name: all_virtual_keys + in: query + description: When true, include clients that are open to all virtual keys (allow_on_all_virtual_keys). ORs with virtual_keys. + schema: + type: boolean + - name: virtual_keys + in: query + description: Comma-separated virtual key IDs; includes clients explicitly assigned to any of them. ORs with all_virtual_keys. + schema: + type: string + - name: code_mode + in: query + description: Filter by code-mode clients. Omit for no filter. + schema: + type: boolean + - name: disabled + in: query + description: Filter by disabled status — true returns disabled clients, false returns enabled clients. Omit for no filter. + schema: + type: boolean security: - ManagementBearerAuth: [] responses: @@ -87,9 +157,9 @@ clients: content: application/json: schema: - type: array - items: - $ref: '../../schemas/management/mcp.yaml#/MCPClient' + $ref: '../../schemas/management/mcp.yaml#/MCPClientsListResponse' + '400': + $ref: '../../openapi.yaml#/components/responses/BadRequest' '500': $ref: '../../openapi.yaml#/components/responses/InternalError' diff --git a/docs/openapi/schemas/inference/chat.yaml b/docs/openapi/schemas/inference/chat.yaml index 2fb42b2484e..0e94352b399 100644 --- a/docs/openapi/schemas/inference/chat.yaml +++ b/docs/openapi/schemas/inference/chat.yaml @@ -97,7 +97,7 @@ ChatCompletionRequest: $ref: '#/ChatPrediction' prompt_cache_retention: type: string - enum: [in-memory, 24h] + enum: [in_memory, 24h] description: Prompt cache retention policy web_search_options: $ref: '#/ChatWebSearchOptions' diff --git a/docs/openapi/schemas/inference/common.yaml b/docs/openapi/schemas/inference/common.yaml index 7b0ab29da92..03d8659b0f4 100644 --- a/docs/openapi/schemas/inference/common.yaml +++ b/docs/openapi/schemas/inference/common.yaml @@ -19,6 +19,7 @@ ModelProvider: - perplexity - replicate - cerebras + - deepseek - gemini - openrouter - elevenlabs diff --git a/docs/openapi/schemas/management/config.yaml b/docs/openapi/schemas/management/config.yaml index d2f3eca9c8b..837483f3bb1 100644 --- a/docs/openapi/schemas/management/config.yaml +++ b/docs/openapi/schemas/management/config.yaml @@ -91,7 +91,7 @@ ClientConfig: items: type: string description: Headers to capture in log metadata. Values are extracted from incoming requests and stored in the metadata field of log entries. Case-insensitive matching. No restart required. - mcp_external_base_url: + mcp_external_client_url: oneOf: - type: string description: Plain URL or env var reference (e.g. "env.MY_VAR") @@ -105,9 +105,61 @@ ClientConfig: type: boolean additionalProperties: false description: > - Public base URL for OAuth callbacks and discovery metadata when Bifrost runs behind a - reverse proxy. Overrides the host derived from the incoming request. Supports env var - syntax ("env.MY_VAR"). Configurable via UI, API, or config.json. + Public base URL Bifrost uses as the redirect_uri when acting as an OAuth client to + upstream MCP servers (Notion, Jira, etc.). Set when Bifrost's callback endpoint is + reached via a different URL than its server-side metadata. Supports env var syntax + ("env.MY_VAR"). + mcp_server_auth_mode: + type: string + enum: [headers, both, oauth] + default: headers + description: > + How /mcp authenticates inbound MCP clients. 'headers' (default): VK/api-key/session + headers only, discovery disabled. 'both': accepts header credentials and Bifrost-issued + JWTs, discovery enabled. 'oauth': Bifrost JWTs only — disables VK/header MCP access. + oauth2_server_config: + type: object + description: > + OAuth2 authorization server settings for /mcp. Only relevant when mcp_server_auth_mode + is 'both' or 'oauth'. + properties: + issuer_url: + oneOf: + - type: string + description: Plain URL or env var reference (e.g. "env.MY_VAR") + - type: object + properties: + value: + type: string + env_var: + type: string + from_env: + type: boolean + additionalProperties: false + description: > + Stable public URL advertised as the OAuth2 AS issuer in discovery documents and the + JWT iss claim. Required for multi-host deployments; single-host can omit it (falls + back to the request Host header). Supports env var syntax ("env.MY_VAR"). + auth_code_ttl: + type: integer + minimum: 1 + maximum: 900 + default: 300 + description: Lifetime of the single-use authorization code in seconds (default 300, max 900 = 15 minutes). + access_token_ttl: + type: integer + minimum: 1 + default: 600 + description: Lifetime of the issued JWT Bearer token in seconds (default 600). + disable_vk_identity: + type: boolean + default: false + description: > + Require identity-provider login: virtual-key identity is removed from the consent + flow, and existing virtual-key-mode grants are rejected at /mcp and denied on refresh. + Only meaningful when mcp_server_auth_mode is 'oauth' and an identity provider is + configured. Anonymous session identity is governed separately by enforce_auth_on_inference. + additionalProperties: false FrameworkConfig: type: object diff --git a/docs/openapi/schemas/management/governance.yaml b/docs/openapi/schemas/management/governance.yaml index 4c6d1e02a47..7237f2e7985 100644 --- a/docs/openapi/schemas/management/governance.yaml +++ b/docs/openapi/schemas/management/governance.yaml @@ -259,6 +259,10 @@ VirtualKey: type: string is_active: type: boolean + expires_at: + type: [string, 'null'] + format: date-time + description: Expiry timestamp. Requests using this virtual key are rejected once it passes. Null or absent means the key never expires. provider_configs: type: array items: @@ -366,6 +370,10 @@ CreateVirtualKeyRequest: calendar_aligned: type: boolean default: false + expires_at: + type: string + format: date-time + description: Optional expiry timestamp. Must be in the future. Omit for a key that never expires. UpdateVirtualKeyRequest: type: object @@ -437,6 +445,9 @@ UpdateVirtualKeyRequest: reset_budget_usage: type: boolean description: When true, resets usage for virtual-key and provider-config budget records reconciled by this update. + expires_at: + type: string + description: RFC3339 timestamp to set a new expiry (must be in the future), empty string to clear an existing expiry, omitted to leave it unchanged. ListVirtualKeysResponse: type: object diff --git a/docs/openapi/schemas/management/mcp.yaml b/docs/openapi/schemas/management/mcp.yaml index a44cc691bf5..5f558b716f9 100644 --- a/docs/openapi/schemas/management/mcp.yaml +++ b/docs/openapi/schemas/management/mcp.yaml @@ -530,6 +530,34 @@ MCPClient: $ref: '#/MCPVKConfigResponse' description: Virtual key assignments for this MCP client +MCPClientsListResponse: + type: object + description: Paginated list of MCP clients. + required: + - clients + - count + - total_count + - limit + - offset + properties: + clients: + type: array + items: + $ref: '#/MCPClient' + count: + type: integer + description: Number of clients returned in this page + total_count: + type: integer + format: int64 + description: Total number of clients matching the query (before pagination) + limit: + type: integer + description: Page size used for the response + offset: + type: integer + description: Page offset used for the response + ExecuteToolRequest: oneOf: - title: Chat (Default) diff --git a/docs/overview.mdx b/docs/overview.mdx index 25b91654f3c..1ce33f4ccdc 100644 --- a/docs/overview.mdx +++ b/docs/overview.mdx @@ -203,6 +203,9 @@ Bifrost supports 20+ AI providers through a single unified API. Configure multip High-speed inference with full streaming support. + + OpenAI-compatible chat, reasoning, tools, and FIM completions. + Local inference with OpenAI-compatible format. diff --git a/docs/providers/provider-routing.mdx b/docs/providers/provider-routing.mdx index 6e1d9960d82..aa0bc2385c3 100644 --- a/docs/providers/provider-routing.mdx +++ b/docs/providers/provider-routing.mdx @@ -1059,7 +1059,7 @@ flowchart TB Cat["Model Catalog Lookup"] Providers["Candidate Providers:
openai, azure, groq"] Filter["Filter by allowed_models
and key availability"] - Score["Score by performance:
error rate, latency, utilization"] + Score["Score by performance:
error rate, latency"] Select["Select: openai"] end @@ -1075,7 +1075,7 @@ flowchart TB ### Level 1: Direction (Provider Selection) -**When it runs**: Only when the model string has **no** provider prefix (e.g., `gpt-4o`) +**When it runs**: Only when no provider has been selected yet — i.e. the request has no `provider/` prefix and no earlier layer (governance, routing rules) pinned one **How it works**: @@ -1083,13 +1083,11 @@ flowchart TB 2. **Provider filtering**: Filter based on: - Allowed models from keys configuration - Keys availability for the provider -3. **Performance scoring**: Calculate scores for each provider based on: - - Error rates (50% weight) - - Latency (20% weight, using MV-TACOS algorithm) - - Utilization (5% weight) - - Momentum bias (recovery acceleration) -4. **Smart selection**: Choose provider using weighted random with jitter and exploration -5. **Fallbacks created**: Remaining providers sorted by performance score (descending) are added as fallbacks +3. **Performance scoring**: Score each provider on its recent, realized performance for the model: + - **Error rate** — the primary, time-decayed signal + - **Token-aware latency** — secondary, comparing the provider both to its peers for the model and to its own recent baseline +4. **Smooth selection**: Concentrate traffic on the best-scoring providers while keeping a small exploration share for the rest, so a recovered provider keeps getting re-probed +5. **Fallbacks created**: Remaining healthy providers sorted by performance score (descending) are added as fallbacks ### Level 2: Route (Key Selection) @@ -1104,21 +1102,23 @@ flowchart TB - Latency (response time) - TPM hits (rate limit violations) - Current state (Healthy, Degraded, Failed, Recovering) -4. **Weighted random selection**: Choose key with exploration (25% chance to probe recovering keys) +4. **Smooth weighted selection**: Concentrate traffic on higher-weight keys, with a small dedicated probe budget reserved for recovering keys so they can prove recovery 5. **Circuit breaker**: Skip keys with zero weight (TPM hits, repeated failures) -### Scoring Algorithm +### Scoring -The load balancer computes a performance score for each provider-model combination: +Every 5 seconds the load balancer recomputes a weight for each route from its recent performance. Three signals drive the score, in priority order: -$$ -Score = (P_{error} \times 0.5) + (P_{latency} \times 0.2) + (P_{util} \times 0.05) - M_{momentum} -$$ +- **Error rate** — the primary, time-decayed signal +- **Token-aware latency** — secondary; a route is compared both to its peers and to its own recent baseline +- **Utilization** — a minor fair-share nudge that discourages overloading any single key + +Which signals apply depends on the route's health: healthy routes are scored mainly on errors and latency, while routes that are actively recovering are scored on latency and recovery progress so they aren't held back by stale error history. Provider-level (Level 1) selection scores on error rate and latency only. - Lower penalties = Higher weights = More traffic. The system self-heals by - quickly penalizing failing routes but enabling fast recovery once issues are - resolved. + Lower penalties = higher weights = more traffic. The system self-heals by + quickly penalizing failing routes but decaying those penalties fast once issues + resolve, so a recovered route returns to full traffic within seconds. ### Request Flow @@ -1139,7 +1139,7 @@ $$ - Groq: Score 0.65 (high latency recently) - OpenAI selected (highest score within jitter band) + OpenAI selected (highest performance score; a small exploration share is still kept for the others) ```json @@ -1161,7 +1161,7 @@ $$ | **Circuit Breakers** | Failing routes automatically removed from rotation | | **Fast Recovery** | 90% penalty reduction in 30 seconds after issues resolve | | **Health States** | Routes transition between Healthy, Degraded, Failed, and Recovering | -| **Smart Exploration** | 25% chance to probe potentially recovered routes | +| **Smart Exploration** | Keeps a small, dedicated share of traffic on recovering routes so they can prove recovery | ### Dashboard Visibility @@ -1181,6 +1181,10 @@ The dashboard shows: - State transitions (Healthy → Degraded → Failed → Recovering) - Actual vs expected traffic distribution + + **Scope & tuning**: Adaptive load balancing operates **per node** (each node routes on its own observed metrics; only rate-limit/TPM backoffs are shared across nodes, and only within a region) and adapts on a **~5-second cycle** — distinct from per-request fallback, which is immediate. Both levels can be toggled independently, and the scoring parameters ship pre-tuned. See [Adaptive Load Balancing](/enterprise/adaptive-load-balancing) for configuration and limitations. + + --- ## How Governance and Load Balancing Interact diff --git a/docs/providers/supported-providers/anthropic.mdx b/docs/providers/supported-providers/anthropic.mdx index 115661e95a1..49331c2e654 100644 --- a/docs/providers/supported-providers/anthropic.mdx +++ b/docs/providers/supported-providers/anthropic.mdx @@ -80,7 +80,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.Anthropic: return []schemas.Key{{ Name: "anthropic-key-1", - Value: *schemas.NewEnvVar("env.ANTHROPIC_API_KEY"), + Value: *schemas.NewSecretVar("env.ANTHROPIC_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/azure.mdx b/docs/providers/supported-providers/azure.mdx index ffa8f5aaa08..d1016d3e26b 100644 --- a/docs/providers/supported-providers/azure.mdx +++ b/docs/providers/supported-providers/azure.mdx @@ -179,7 +179,7 @@ func (a *MyAccount) GetKeysForProvider(ctx *context.Context, provider schemas.Mo "gpt-4o-mini": "my-mini-deployment", }, AzureKeyConfig: &schemas.AzureKeyConfig{ - Endpoint: *schemas.NewEnvVar(os.Getenv("AZURE_ENDPOINT")), + Endpoint: *schemas.NewSecretVar(os.Getenv("AZURE_ENDPOINT")), }, }, }, nil @@ -315,10 +315,10 @@ func (a *MyAccount) GetKeysForProvider(ctx *context.Context, provider schemas.Mo "claude-3-5-sonnet": "my-claude-deployment", }, AzureKeyConfig: &schemas.AzureKeyConfig{ - Endpoint: *schemas.NewEnvVar(os.Getenv("AZURE_ENDPOINT")), - ClientID: schemas.NewEnvVar(os.Getenv("AZURE_CLIENT_ID")), - ClientSecret: schemas.NewEnvVar(os.Getenv("AZURE_CLIENT_SECRET")), - TenantID: schemas.NewEnvVar(os.Getenv("AZURE_TENANT_ID")), + Endpoint: *schemas.NewSecretVar(os.Getenv("AZURE_ENDPOINT")), + ClientID: schemas.NewSecretVar(os.Getenv("AZURE_CLIENT_ID")), + ClientSecret: schemas.NewSecretVar(os.Getenv("AZURE_CLIENT_SECRET")), + TenantID: schemas.NewSecretVar(os.Getenv("AZURE_TENANT_ID")), Scopes: []string{"https://cognitiveservices.azure.com/.default"}, }, }, @@ -438,7 +438,7 @@ func (a *MyAccount) GetKeysForProvider(ctx *context.Context, provider schemas.Mo case schemas.Azure: return []schemas.Key{ { - Value: *schemas.NewEnvVar("env.AZURE_OPENAI_KEY"), + Value: *schemas.NewSecretVar("env.AZURE_OPENAI_KEY"), Models: []string{"*"}, Weight: 1.0, Aliases: schemas.KeyAliases{ @@ -446,7 +446,7 @@ func (a *MyAccount) GetKeysForProvider(ctx *context.Context, provider schemas.Mo "gpt-4o-mini": "my-mini-deployment", }, AzureKeyConfig: &schemas.AzureKeyConfig{ - Endpoint: *schemas.NewEnvVar(os.Getenv("AZURE_ENDPOINT")), + Endpoint: *schemas.NewSecretVar(os.Getenv("AZURE_ENDPOINT")), }, }, }, nil diff --git a/docs/providers/supported-providers/bedrock-mantle.mdx b/docs/providers/supported-providers/bedrock-mantle.mdx new file mode 100644 index 00000000000..9fef226e33b --- /dev/null +++ b/docs/providers/supported-providers/bedrock-mantle.mdx @@ -0,0 +1,166 @@ +--- +title: "AWS Bedrock Mantle" +description: "AWS Bedrock Mantle provider - a single endpoint serving Claude (native Anthropic Messages) and OpenAI-family / Gemma models (OpenAI-compatible), with AWS SigV4 or API-key auth" +icon: "aws" +--- + +## Overview + +**Bedrock Mantle** is a single AWS endpoint, served on the `bedrock-mantle.{region}.api.aws` host, that exposes a broad model catalog through two surfaces: + +- **Native Anthropic Messages API** (`/anthropic/v1/messages`) for Claude models. +- **OpenAI-compatible API** (`/v1/...` or `/openai/v1/...`) for OpenAI-family (`gpt-*`), Gemma, and other open models. + +Bifrost dispatches each request to the correct surface automatically based on the model family — you address the provider the same way regardless: `bedrock_mantle/`. + + +Bedrock Mantle is distinct from the [Bedrock](./bedrock) provider, which uses the Converse / InvokeModel APIs on `bedrock-runtime`. Mantle is its own provider with its own credentials block (`bedrock_mantle_key_config`). + + +### Model IDs + +Mantle model IDs are sent **verbatim** to the endpoint and differ from the Converse IDs used by the Bedrock provider — there is **no cross-region prefix** (`global.`/`us.`) and **no version suffix** (`-v1:0`): + +| Surface | Format | Examples | +| --- | --- | --- | +| Native Anthropic (Claude) | `anthropic.{model}` | `anthropic.claude-opus-4-8`, `anthropic.claude-haiku-4-5` | +| OpenAI-compatible | `openai.{model}` / `google.{model}` / … | `openai.gpt-oss-120b`, `openai.gpt-5.5`, `google.gemma-4-31b` | + +You can discover the exact IDs your account serves with a **list-models** request (the `/v1/models` catalog). An optional leading `region/` prefix may be used to address a specific region per request (e.g. `bedrock_mantle/us-west-2/anthropic.claude-opus-4-8`). + +### Supported Operations + +| Operation | Supported | +| --- | --- | +| Chat Completions (+ streaming) | ✅ | +| Responses API (+ streaming) | ✅ | +| Tool calling (+ streaming) | ✅ | +| Vision (image input, Claude) | ✅ | +| Reasoning / extended thinking (Claude) | ✅ | +| Structured outputs | ✅ | +| Prompt caching (Claude) | ✅ | +| List models | ✅ | +| Text completion, embeddings, batch, files, images, audio, count tokens, rerank | ❌ | + +For Claude models, request/response handling (beta headers, cache control, reasoning, message conversion) follows the same rules as the [Bedrock provider's Anthropic behavior](./bedrock) — refer to that page for the deep parameter mapping. + +--- + +## Setup & Configuration + +Bedrock Mantle authenticates with **AWS SigV4** (the `bedrock-mantle` signing service) or an optional **Bearer API key**. A `region` is **required** on the key config (there is no default). + +### 1. SigV4 — Explicit Credentials + + +```json config.json +{ + "providers": { + "bedrock_mantle": { + "keys": [ + { + "name": "mantle-static", + "value": "", + "models": ["*"], + "weight": 1.0, + "bedrock_mantle_key_config": { + "region": "us-east-1", + "access_key": "env.AWS_ACCESS_KEY_ID", + "secret_key": "env.AWS_SECRET_ACCESS_KEY", + "session_token": "env.AWS_SESSION_TOKEN" + } + } + ] + } + } +} +``` + + +`session_token` is optional (only needed for temporary credentials). + +### 2. SigV4 — Inherited AWS Credentials / IAM Role + +Leave `access_key` and `secret_key` empty and Bifrost inherits credentials from the AWS SDK default chain — IRSA (IAM Roles for Service Accounts), EC2 instance profile, or `AWS_*` environment variables. Only `region` is required. + + +```json config.json +{ + "providers": { + "bedrock_mantle": { + "keys": [ + { + "name": "mantle-iam", + "value": "", + "models": ["*"], + "weight": 1.0, + "bedrock_mantle_key_config": { + "region": "us-east-1" + } + } + ] + } + } +} +``` + + +To assume an IAM role before requests (works with both explicit and inherited credentials), add `role_arn` (and optionally `external_id` / `session_name`): + + +```json config.json +{ + "bedrock_mantle_key_config": { + "region": "us-west-2", + "role_arn": "env.AWS_ROLE_ARN", + "external_id": "env.AWS_EXTERNAL_ID", + "session_name": "bifrost-session" + } +} +``` + + +### 3. API Key (Bearer) + +Alternatively, authenticate with a Bedrock Mantle API key sent as a Bearer token. Set the top-level key `value` and leave the SigV4 credentials empty (`region` is still required): + + +```json config.json +{ + "providers": { + "bedrock_mantle": { + "keys": [ + { + "name": "mantle-api-key", + "value": "env.BEDROCK_MANTLE_API_KEY", + "models": ["*"], + "weight": 1.0, + "bedrock_mantle_key_config": { + "region": "us-east-1" + } + } + ] + } + } +} +``` + + +--- + +## Usage + +Address models with the `bedrock_mantle/` prefix: + + +```bash cURL +curl -X POST http://localhost:8080/v1/chat/completions \ + -H "Content-Type: application/json" \ + -d '{ + "model": "bedrock_mantle/anthropic.claude-opus-4-8", + "messages": [{ "role": "user", "content": "Hello!" }] + }' +``` + + +Both surfaces are addressed identically — `bedrock_mantle/anthropic.claude-opus-4-8` (native Anthropic) and `bedrock_mantle/openai.gpt-oss-120b` (OpenAI-compatible) — Bifrost routes each to the right API. diff --git a/docs/providers/supported-providers/bedrock.mdx b/docs/providers/supported-providers/bedrock.mdx index e03ee3a2fbd..d5504de32be 100644 --- a/docs/providers/supported-providers/bedrock.mdx +++ b/docs/providers/supported-providers/bedrock.mdx @@ -189,10 +189,10 @@ func (a *MyAccount) GetKeysForProvider(ctx *context.Context, provider schemas.Mo "claude-3-5-sonnet": "us.anthropic.claude-3-5-sonnet-20241022-v2:0", }, BedrockKeyConfig: &schemas.BedrockKeyConfig{ - AccessKey: *schemas.NewEnvVar("env.AWS_ACCESS_KEY_ID"), - SecretKey: *schemas.NewEnvVar("env.AWS_SECRET_ACCESS_KEY"), - SessionToken: schemas.NewEnvVar("env.AWS_SESSION_TOKEN"), - Region: schemas.NewEnvVar("us-east-1"), + AccessKey: *schemas.NewSecretVar("env.AWS_ACCESS_KEY_ID"), + SecretKey: *schemas.NewSecretVar("env.AWS_SECRET_ACCESS_KEY"), + SessionToken: schemas.NewSecretVar("env.AWS_SESSION_TOKEN"), + Region: schemas.NewSecretVar("us-east-1"), }, }, }, nil @@ -331,10 +331,10 @@ func (a *MyAccount) GetKeysForProvider(ctx *context.Context, provider schemas.Mo }, BedrockKeyConfig: &schemas.BedrockKeyConfig{ // Leave AccessKey and SecretKey empty - resolved from IRSA/instance profile/env vars - Region: schemas.NewEnvVar("us-east-1"), - RoleARN: schemas.NewEnvVar("env.AWS_ROLE_ARN"), // optional - ExternalID: schemas.NewEnvVar("env.AWS_EXTERNAL_ID"), // optional - RoleSessionName: schemas.NewEnvVar("bifrost-session"), // optional + Region: schemas.NewSecretVar("us-east-1"), + RoleARN: schemas.NewSecretVar("env.AWS_ROLE_ARN"), // optional + ExternalID: schemas.NewSecretVar("env.AWS_EXTERNAL_ID"), // optional + RoleSessionName: schemas.NewSecretVar("bifrost-session"), // optional }, }, }, nil @@ -428,11 +428,11 @@ func (a *MyAccount) GetKeysForProvider(ctx *context.Context, provider schemas.Mo case schemas.Bedrock: return []schemas.Key{ { - Value: *schemas.NewEnvVar("env.BEDROCK_API_KEY"), + Value: *schemas.NewSecretVar("env.BEDROCK_API_KEY"), Models: []string{"*"}, Weight: 1.0, BedrockKeyConfig: &schemas.BedrockKeyConfig{ - Region: schemas.NewEnvVar("us-east-1"), + Region: schemas.NewSecretVar("us-east-1"), }, }, }, nil diff --git a/docs/providers/supported-providers/cerebras.mdx b/docs/providers/supported-providers/cerebras.mdx index 2b06ddc7805..49c3bc517d1 100644 --- a/docs/providers/supported-providers/cerebras.mdx +++ b/docs/providers/supported-providers/cerebras.mdx @@ -80,7 +80,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.Cerebras: return []schemas.Key{{ Name: "cerebras-key-1", - Value: *schemas.NewEnvVar("env.CEREBRAS_API_KEY"), + Value: *schemas.NewSecretVar("env.CEREBRAS_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/cohere.mdx b/docs/providers/supported-providers/cohere.mdx index 4f596f2b3bb..359faab581e 100644 --- a/docs/providers/supported-providers/cohere.mdx +++ b/docs/providers/supported-providers/cohere.mdx @@ -80,7 +80,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.Cohere: return []schemas.Key{{ Name: "cohere-key-1", - Value: *schemas.NewEnvVar("env.COHERE_API_KEY"), + Value: *schemas.NewSecretVar("env.COHERE_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/deepseek.mdx b/docs/providers/supported-providers/deepseek.mdx new file mode 100644 index 00000000000..b37729590a9 --- /dev/null +++ b/docs/providers/supported-providers/deepseek.mdx @@ -0,0 +1,215 @@ +--- +title: "DeepSeek" +description: "DeepSeek API conversion guide - OpenAI-compatible format, chat, streaming, tool calling, reasoning, and beta text completions" +icon: "d" +--- + +## Overview + +DeepSeek is an **OpenAI-compatible provider** with a dedicated Bifrost provider implementation for DeepSeek's endpoint layout. Bifrost uses the shared OpenAI-compatible request and response converters, while preserving DeepSeek-specific extra parameters and routing text completions to the beta FIM endpoint. Key characteristics: + +- **OpenAI-compatible chat** - Chat Completions use `/chat/completions` +- **Streaming support** - Server-Sent Events for chat and text completions +- **Tool calling** - Function tools are passed through using the OpenAI-compatible schema +- **Reasoning support** - Reasoning parameters and DeepSeek extra parameters are forwarded +- **Responses API** - Supported by converting Responses requests to Chat Completions internally +- **Beta text completions** - Text/FIM completions use DeepSeek's `/beta/completions` endpoint + +### Supported Operations + +| Operation | Non-Streaming | Streaming | Endpoint | +|-----------|---------------|-----------|----------| +| Chat Completions | ✅ | ✅ | `/chat/completions` | +| Responses API | ✅ | ✅ | `/chat/completions` | +| Text Completions | ✅ | ✅ | `/beta/completions` | +| List Models | ✅ | - | `/models` | +| Embeddings | ❌ | ❌ | - | +| Image Generation | ❌ | ❌ | - | +| Speech (TTS) | ❌ | ❌ | - | +| Transcriptions (STT) | ❌ | ❌ | - | +| Files | ❌ | ❌ | - | +| Batch | ❌ | ❌ | - | + + +**Unsupported Operations** (❌): Embeddings, Image Generation, Speech, Transcriptions, Files, Batch, cached content, containers, token counting, compaction, OCR, rerank, video, and passthrough are not supported by the upstream DeepSeek API through this provider. These return `UnsupportedOperationError`. + + +## Setup & Configuration + +Configure DeepSeek as a provider. + + + + +![DeepSeek provider dashboard](../../media/provider-dashboard-deepseek.png) + +1. Navigate to **Models** > **Model Providers**. Look for **DeepSeek** under **Configured Providers**. If it is missing, click on **Add New Provider** and select **DeepSeek**. +2. Click **Add Key** or edit an existing key. +3. Set a name for your key. +4. Paste your API key directly or use an environment variable (for example, `env.DEEPSEEK_API_KEY`). +5. Set **Allowed Models** to **All Models** (default) or the specific model allowlist you want this key to serve. +6. Save the provider configuration. + + + + +```json +{ + "providers": { + "deepseek": { + "keys": [ + { + "name": "deepseek-key-1", + "value": "env.DEEPSEEK_API_KEY", + "models": [ + "*" + ], + "weight": 1.0 + } + ] + } + } +} +``` + + + +Refer to the API documentation for [Provider Keys Management](https://docs.getbifrost.ai/api-reference/providers/create-a-key-for-a-provider). + + + +```go +case schemas.DeepSeek: + return []schemas.Key{{ + Name: "deepseek-key-1", + Value: *schemas.NewSecretVar("env.DEEPSEEK_API_KEY"), + Models: []string{"*"}, + Weight: 1.0, + }}, nil +``` + + + + +--- + +# 1. Chat Completions + +## Request Parameters + +DeepSeek supports OpenAI-compatible chat completion parameters. For the full parameter reference and message conversion behavior, see [OpenAI Chat Completions](/providers/supported-providers/openai#1-chat-completions). + +### Filtered Parameters + +Removed for DeepSeek compatibility: +- `prediction` - OpenAI-specific predicted output +- `prompt_cache_key` - OpenAI-specific prompt cache key +- `prompt_cache_retention` - OpenAI-specific prompt cache retention +- `verbosity` - Anthropic-specific +- `store` - OpenAI-specific response storage +- `web_search_options` - OpenAI-specific web search options + +### Reasoning Parameter + +DeepSeek delegates through `ToOpenAIChatRequest` with provider-specific compatibility handling. Reasoning effort is normalized using the OpenAI-compatible provider convention, and DeepSeek V4 models preserve `reasoning.effort: "max"` when requested. + +Assistant-message `reasoning` details are stripped before sending follow-up messages because DeepSeek rejects `reasoning_details` in assistant messages. + +### Extra Parameters + +DeepSeek enables passthrough extra parameters for chat and text completion requests. Provider-specific options such as DeepSeek thinking controls can be sent through `extra_params` without being dropped by Bifrost. + +DeepSeek supports standard OpenAI message types, tools, responses, and streaming formats. For details on message handling, tool conversion, responses, and streaming, refer to [OpenAI Chat Completions](/providers/supported-providers/openai#1-chat-completions). + +--- + +# 2. Responses API + +Bifrost converts Responses API format to Chat Completions internally, then converts the response back: + +``` +BifrostResponsesRequest + → ToChatRequest() + → ChatCompletion + → ToBifrostResponsesResponse() +``` + +Same parameter support as Chat Completions with response format differences (output items instead of message content). Streaming Responses requests are also routed through Chat Completions streaming. + +--- + +# 3. Text Completions + +DeepSeek supports beta text/FIM completions through `/beta/completions`: + +| Parameter | Mapping | +|-----------|---------| +| `prompt` | Sent as-is | +| `max_tokens` | max_tokens | +| `temperature` | temperature | +| `top_p` | top_p | +| `stop` | stop sequences | +| `extra_params` | Passed through to DeepSeek | + +Response returns `choices[].text` with completion text. + +--- + +# 4. Text Completions Streaming + +Streaming text completions use DeepSeek's OpenAI-compatible SSE format on `/beta/completions`. + +--- + +# 5. List Models + +Lists available models from DeepSeek through `/models`. + +--- + +## Unsupported Features + +| Feature | Reason | +|---------|--------| +| Embedding | Not offered by DeepSeek API through this provider | +| Image Generation | Not offered by DeepSeek API through this provider | +| Speech/TTS | Not offered by DeepSeek API through this provider | +| Transcription/STT | Not offered by DeepSeek API through this provider | +| Batch Operations | Not offered by DeepSeek API through this provider | +| File Management | Not offered by DeepSeek API through this provider | +| Cached Content | Only Gemini and Vertex AI support cached content in Bifrost | +| Container Management | Not offered by DeepSeek API through this provider | +| Token Counting | Not offered by DeepSeek API through this provider | +| Rerank/OCR/Video | Not offered by DeepSeek API through this provider | + +--- + +## Caveats + + +**Severity**: Low +**Behavior**: DeepSeek defaults to `https://api.deepseek.com` +**Impact**: Custom DeepSeek-compatible deployments must override `network_config.base_url` +**Code**: `NewDeepSeekProvider` sets the default base URL when no provider-level base URL is configured + + + +**Severity**: Medium +**Behavior**: Text completions are routed to `/beta/completions` +**Impact**: FIM/text completion behavior follows DeepSeek's beta API contract and may differ from standard OpenAI `/completions` +**Code**: `TextCompletion` and `TextCompletionStream` use `/beta/completions` + + + +**Severity**: Low +**Behavior**: User field > 64 characters is silently dropped +**Impact**: Longer user identifiers are lost +**Code**: `SanitizeUserField` enforces 64-char max in the shared OpenAI converter + + + +**Severity**: Medium +**Behavior**: Assistant-message `reasoning` details are removed before forwarding follow-up chat messages +**Impact**: Prevents DeepSeek request failures when previous assistant turns contain reasoning metadata +**Code**: `stripReasoningDetails` applies to DeepSeek in `ToOpenAIChatRequest` + diff --git a/docs/providers/supported-providers/elevenlabs.mdx b/docs/providers/supported-providers/elevenlabs.mdx index f286d4388da..f107097e9db 100644 --- a/docs/providers/supported-providers/elevenlabs.mdx +++ b/docs/providers/supported-providers/elevenlabs.mdx @@ -82,7 +82,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.Elevenlabs: return []schemas.Key{{ Name: "elevenlabs-key-1", - Value: *schemas.NewEnvVar("env.ELEVENLABS_API_KEY"), + Value: *schemas.NewSecretVar("env.ELEVENLABS_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/fireworks.mdx b/docs/providers/supported-providers/fireworks.mdx index 4220f9fe953..56823b6831a 100644 --- a/docs/providers/supported-providers/fireworks.mdx +++ b/docs/providers/supported-providers/fireworks.mdx @@ -83,7 +83,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.Fireworks: return []schemas.Key{{ Name: "fireworks-key-1", - Value: *schemas.NewEnvVar("env.FIREWORKS_API_KEY"), + Value: *schemas.NewSecretVar("env.FIREWORKS_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/gemini.mdx b/docs/providers/supported-providers/gemini.mdx index e2c910ee761..0881dfc87a9 100644 --- a/docs/providers/supported-providers/gemini.mdx +++ b/docs/providers/supported-providers/gemini.mdx @@ -81,7 +81,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.Gemini: return []schemas.Key{{ Name: "gemini-key-1", - Value: *schemas.NewEnvVar("env.GEMINI_API_KEY"), + Value: *schemas.NewSecretVar("env.GEMINI_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/groq.mdx b/docs/providers/supported-providers/groq.mdx index 9bb52c7801e..edbd0da7678 100644 --- a/docs/providers/supported-providers/groq.mdx +++ b/docs/providers/supported-providers/groq.mdx @@ -82,7 +82,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.Groq: return []schemas.Key{{ Name: "groq-key-1", - Value: *schemas.NewEnvVar("env.GROQ_API_KEY"), + Value: *schemas.NewSecretVar("env.GROQ_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/huggingface.mdx b/docs/providers/supported-providers/huggingface.mdx index 396c54ac53f..d4d1650819f 100644 --- a/docs/providers/supported-providers/huggingface.mdx +++ b/docs/providers/supported-providers/huggingface.mdx @@ -91,7 +91,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.HuggingFace: return []schemas.Key{{ Name: "huggingface-key-1", - Value: *schemas.NewEnvVar("env.HUGGINGFACE_API_KEY"), + Value: *schemas.NewSecretVar("env.HUGGINGFACE_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/mistral.mdx b/docs/providers/supported-providers/mistral.mdx index e48833a66b6..dee8d6bf007 100644 --- a/docs/providers/supported-providers/mistral.mdx +++ b/docs/providers/supported-providers/mistral.mdx @@ -87,7 +87,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.Mistral: return []schemas.Key{{ Name: "mistral-key-1", - Value: *schemas.NewEnvVar("env.MISTRAL_API_KEY"), + Value: *schemas.NewSecretVar("env.MISTRAL_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/nebius.mdx b/docs/providers/supported-providers/nebius.mdx index dad778d5a22..e34fccd214f 100644 --- a/docs/providers/supported-providers/nebius.mdx +++ b/docs/providers/supported-providers/nebius.mdx @@ -80,7 +80,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.Nebius: return []schemas.Key{{ Name: "nebius-key-1", - Value: *schemas.NewEnvVar("env.NEBIUS_API_KEY"), + Value: *schemas.NewSecretVar("env.NEBIUS_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/ollama.mdx b/docs/providers/supported-providers/ollama.mdx index b43e3c2cfd8..9ff1de182cd 100644 --- a/docs/providers/supported-providers/ollama.mdx +++ b/docs/providers/supported-providers/ollama.mdx @@ -169,11 +169,11 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.Ollama: return []schemas.Key{{ Name: "ollama-local", - Value: *schemas.NewEnvVar(""), + Value: *schemas.NewSecretVar(""), Models: []string{"*"}, Weight: 1.0, OllamaKeyConfig: &schemas.OllamaKeyConfig{ - URL: *schemas.NewEnvVar("http://localhost:11434"), + URL: *schemas.NewSecretVar("http://localhost:11434"), }, }}, nil ``` diff --git a/docs/providers/supported-providers/openai.mdx b/docs/providers/supported-providers/openai.mdx index 3a4d7075ef5..b9cffb81091 100644 --- a/docs/providers/supported-providers/openai.mdx +++ b/docs/providers/supported-providers/openai.mdx @@ -78,7 +78,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.OpenAI: return []schemas.Key{{ Name: "openai-key-1", - Value: *schemas.NewEnvVar("env.OPENAI_API_KEY"), + Value: *schemas.NewSecretVar("env.OPENAI_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/opencode.mdx b/docs/providers/supported-providers/opencode.mdx index c63ed27fbb2..d3b86f5149f 100644 --- a/docs/providers/supported-providers/opencode.mdx +++ b/docs/providers/supported-providers/opencode.mdx @@ -112,7 +112,7 @@ For OpenCode Zen: case schemas.OpencodeZen: return []schemas.Key{{ Name: "zen-key-1", - Value: *schemas.NewEnvVar("env.OPENCODE_API_KEY"), + Value: *schemas.NewSecretVar("env.OPENCODE_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil @@ -124,7 +124,7 @@ For OpenCode Go: case schemas.OpencodeGo: return []schemas.Key{{ Name: "go-key-1", - Value: *schemas.NewEnvVar("env.OPENCODE_API_KEY"), + Value: *schemas.NewSecretVar("env.OPENCODE_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/openrouter.mdx b/docs/providers/supported-providers/openrouter.mdx index e6e2f0949a7..d62a544621c 100644 --- a/docs/providers/supported-providers/openrouter.mdx +++ b/docs/providers/supported-providers/openrouter.mdx @@ -82,7 +82,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.OpenRouter: return []schemas.Key{{ Name: "openrouter-key-1", - Value: *schemas.NewEnvVar("env.OPENROUTER_API_KEY"), + Value: *schemas.NewSecretVar("env.OPENROUTER_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/overview.mdx b/docs/providers/supported-providers/overview.mdx index 6d8cd586c68..2a4b9d998c0 100644 --- a/docs/providers/supported-providers/overview.mdx +++ b/docs/providers/supported-providers/overview.mdx @@ -19,8 +19,10 @@ The following table summarizes which operations are supported by each provider v | Anthropic (`anthropic/`) | ✅ | ✅ | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | ✅ | | Azure (`azure/`) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ | ✅ | ❌ | ❌ | ✅ | ✅ | | Bedrock (`bedrock/`) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ✅ | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | +| Bedrock Mantle (`bedrock_mantle/`) | ✅ | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | | Cerebras (`cerebras/`) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | | Cohere (`cohere/`) | ✅ | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | +| DeepSeek (`deepseek/`) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | | Elevenlabs (`elevenlabs/`) | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | | Fireworks (`fireworks/`) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | | Gemini (`gemini/`) | ✅ | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ✅ | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ✅ | ❌ | ❌ | ✅ | ✅ | diff --git a/docs/providers/supported-providers/parasail.mdx b/docs/providers/supported-providers/parasail.mdx index 8630e94ff5f..c62860838a6 100644 --- a/docs/providers/supported-providers/parasail.mdx +++ b/docs/providers/supported-providers/parasail.mdx @@ -80,7 +80,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.Parasail: return []schemas.Key{{ Name: "parasail-key-1", - Value: *schemas.NewEnvVar("env.PARASAIL_API_KEY"), + Value: *schemas.NewSecretVar("env.PARASAIL_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/perplexity.mdx b/docs/providers/supported-providers/perplexity.mdx index 3e6d36601df..1fbdd94f2bb 100644 --- a/docs/providers/supported-providers/perplexity.mdx +++ b/docs/providers/supported-providers/perplexity.mdx @@ -80,7 +80,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.Perplexity: return []schemas.Key{{ Name: "perplexity-key-1", - Value: *schemas.NewEnvVar("env.PERPLEXITY_API_KEY"), + Value: *schemas.NewSecretVar("env.PERPLEXITY_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/replicate.mdx b/docs/providers/supported-providers/replicate.mdx index 33478fc178b..6c6724f7c5f 100644 --- a/docs/providers/supported-providers/replicate.mdx +++ b/docs/providers/supported-providers/replicate.mdx @@ -89,7 +89,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.Replicate: return []schemas.Key{{ Name: "replicate-key-1", - Value: *schemas.NewEnvVar("env.REPLICATE_API_TOKEN"), + Value: *schemas.NewSecretVar("env.REPLICATE_API_TOKEN"), Models: []string{"*"}, Weight: 1.0, ReplicateKeyConfig: &schemas.ReplicateKeyConfig{ diff --git a/docs/providers/supported-providers/runware.mdx b/docs/providers/supported-providers/runware.mdx index 9706a986436..4e9f890b1dc 100644 --- a/docs/providers/supported-providers/runware.mdx +++ b/docs/providers/supported-providers/runware.mdx @@ -191,7 +191,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.Runware: return []schemas.Key{{ Name: "runware-key-1", - Value: *schemas.NewEnvVar("env.RUNWARE_API_KEY"), + Value: *schemas.NewSecretVar("env.RUNWARE_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/runway.mdx b/docs/providers/supported-providers/runway.mdx index 106c2dfaf6e..54459bbf2b0 100644 --- a/docs/providers/supported-providers/runway.mdx +++ b/docs/providers/supported-providers/runway.mdx @@ -116,7 +116,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.Runway: return []schemas.Key{{ Name: "runway-key-1", - Value: *schemas.NewEnvVar("env.RUNWAY_API_KEY"), + Value: *schemas.NewSecretVar("env.RUNWAY_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/providers/supported-providers/sgl.mdx b/docs/providers/supported-providers/sgl.mdx index 7ff7aa4ad2a..cae41f3c413 100644 --- a/docs/providers/supported-providers/sgl.mdx +++ b/docs/providers/supported-providers/sgl.mdx @@ -86,11 +86,11 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.SGL: return []schemas.Key{{ Name: "sgl-local", - Value: *schemas.NewEnvVar(""), + Value: *schemas.NewSecretVar(""), Models: []string{"*"}, Weight: 1.0, SGLKeyConfig: &schemas.SGLKeyConfig{ - URL: *schemas.NewEnvVar("http://localhost:8000"), + URL: *schemas.NewSecretVar("http://localhost:8000"), }, }}, nil ``` diff --git a/docs/providers/supported-providers/vertex.mdx b/docs/providers/supported-providers/vertex.mdx index aa81b3cd93c..3fe7828ce41 100644 --- a/docs/providers/supported-providers/vertex.mdx +++ b/docs/providers/supported-providers/vertex.mdx @@ -158,9 +158,9 @@ func (a *MyAccount) GetKeysForProvider(ctx *context.Context, provider schemas.Mo Models: []string{"*"}, Weight: 1.0, VertexKeyConfig: &schemas.VertexKeyConfig{ - ProjectID: *schemas.NewEnvVar("env.VERTEX_PROJECT_ID"), - Region: *schemas.NewEnvVar("us-central1"), - AuthCredentials: *schemas.NewEnvVar("env.VERTEX_CREDENTIALS"), // full service account JSON + ProjectID: *schemas.NewSecretVar("env.VERTEX_PROJECT_ID"), + Region: *schemas.NewSecretVar("us-central1"), + AuthCredentials: *schemas.NewSecretVar("env.VERTEX_CREDENTIALS"), // full service account JSON }, }, }, nil @@ -274,8 +274,8 @@ func (a *MyAccount) GetKeysForProvider(ctx *context.Context, provider schemas.Mo Models: []string{"*"}, Weight: 1.0, VertexKeyConfig: &schemas.VertexKeyConfig{ - ProjectID: *schemas.NewEnvVar("env.VERTEX_PROJECT_ID"), - Region: *schemas.NewEnvVar("us-central1"), + ProjectID: *schemas.NewSecretVar("env.VERTEX_PROJECT_ID"), + Region: *schemas.NewSecretVar("us-central1"), // Leave AuthCredentials empty - uses Application Default Credentials }, }, @@ -402,16 +402,16 @@ func (a *MyAccount) GetKeysForProvider(ctx *context.Context, provider schemas.Mo case schemas.Vertex: return []schemas.Key{ { - Value: *schemas.NewEnvVar("env.VERTEX_API_KEY"), // only when using Gemini or fine-tuned models + Value: *schemas.NewSecretVar("env.VERTEX_API_KEY"), // only when using Gemini or fine-tuned models Models: []string{"gemini-pro", "gemini-2.0-flash", "my-fine-tuned-model"}, Weight: 1.0, Aliases: schemas.KeyAliases{ "my-fine-tuned-model": "123456789", }, VertexKeyConfig: &schemas.VertexKeyConfig{ - ProjectID: *schemas.NewEnvVar("env.VERTEX_PROJECT_ID"), - ProjectNumber: *schemas.NewEnvVar("env.VERTEX_PROJECT_NUMBER"), // required for fine-tuned models - Region: *schemas.NewEnvVar("us-central1"), + ProjectID: *schemas.NewSecretVar("env.VERTEX_PROJECT_ID"), + ProjectNumber: *schemas.NewSecretVar("env.VERTEX_PROJECT_NUMBER"), // required for fine-tuned models + Region: *schemas.NewSecretVar("us-central1"), }, }, }, nil diff --git a/docs/providers/supported-providers/vllm.mdx b/docs/providers/supported-providers/vllm.mdx index cd78deeb648..aa2277610bd 100644 --- a/docs/providers/supported-providers/vllm.mdx +++ b/docs/providers/supported-providers/vllm.mdx @@ -87,11 +87,11 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.VLLM: return []schemas.Key{{ Name: "vllm-local", - Value: *schemas.NewEnvVar(""), + Value: *schemas.NewSecretVar(""), Models: []string{"meta-llama/Llama-3.2-1B-Instruct"}, Weight: 1.0, VLLMKeyConfig: &schemas.VLLMKeyConfig{ - URL: *schemas.NewEnvVar("http://localhost:8000"), + URL: *schemas.NewSecretVar("http://localhost:8000"), ModelName: "meta-llama/Llama-3.2-1B-Instruct", }, }}, nil diff --git a/docs/providers/supported-providers/xai.mdx b/docs/providers/supported-providers/xai.mdx index 67610ba0bb4..4de670473f0 100644 --- a/docs/providers/supported-providers/xai.mdx +++ b/docs/providers/supported-providers/xai.mdx @@ -82,7 +82,7 @@ Refer to the API documentation for [Provider Keys Management](https://docs.getbi case schemas.XAI: return []schemas.Key{{ Name: "xai-key-1", - Value: *schemas.NewEnvVar("env.XAI_API_KEY"), + Value: *schemas.NewSecretVar("env.XAI_API_KEY"), Models: []string{"*"}, Weight: 1.0, }}, nil diff --git a/docs/quickstart/gateway/provider-configuration.mdx b/docs/quickstart/gateway/provider-configuration.mdx index 748a537d7f1..d6bb1ba1893 100644 --- a/docs/quickstart/gateway/provider-configuration.mdx +++ b/docs/quickstart/gateway/provider-configuration.mdx @@ -175,6 +175,7 @@ export OPENAI_API_KEY="your-openai-api-key" export ANTHROPIC_API_KEY="your-anthropic-api-key" export MISTRAL_API_KEY="your-mistral-api-key" export CEREBRAS_API_KEY="your-cerebras-api-key" +export DEEPSEEK_API_KEY="your-deepseek-api-key" export GROQ_API_KEY="your-groq-api-key" export COHERE_API_KEY="your-cohere-api-key" ``` diff --git a/docs/quickstart/go-sdk/provider-configuration.mdx b/docs/quickstart/go-sdk/provider-configuration.mdx index eb6cdec1b04..1ac4bc18185 100644 --- a/docs/quickstart/go-sdk/provider-configuration.mdx +++ b/docs/quickstart/go-sdk/provider-configuration.mdx @@ -70,6 +70,7 @@ Set up your API keys for the providers you want to use: export OPENAI_API_KEY="your-openai-api-key" export ANTHROPIC_API_KEY="your-anthropic-api-key" export CEREBRAS_API_KEY="your-cerebras-api-key" +export DEEPSEEK_API_KEY="your-deepseek-api-key" export MISTRAL_API_KEY="your-mistral-api-key" export GROQ_API_KEY="your-groq-api-key" export COHERE_API_KEY="your-cohere-api-key" diff --git a/examples/configs/withclickhouselogstore/config.json b/examples/configs/withclickhouselogstore/config.json new file mode 100644 index 00000000000..c9ae21385a1 --- /dev/null +++ b/examples/configs/withclickhouselogstore/config.json @@ -0,0 +1,34 @@ +{ + "$schema": "https://www.getbifrost.ai/schema", + "config_store": { + "enabled": true, + "type": "sqlite", + "config": { + "path": "../../examples/configs/withclickhouselogstore/config.db" + } + }, + "logs_store": { + "enabled": true, + "type": "clickhouse", + "retention_days": 30, + "config": { + "host": "localhost", + "port": "9001", + "database": "bifrost", + "username": "bifrost", + "password": "bifrost_password" + } + }, + "providers": { + "openai": { + "keys": [ + { + "name": "openai-key-1", + "value": "sk-proj-abc", + "weight": 1, + "models": ["*"] + } + ] + } + } +} diff --git a/examples/configs/withclickhouselogstorehttp/config.json b/examples/configs/withclickhouselogstorehttp/config.json new file mode 100644 index 00000000000..07b6cc8db57 --- /dev/null +++ b/examples/configs/withclickhouselogstorehttp/config.json @@ -0,0 +1,35 @@ +{ + "$schema": "https://www.getbifrost.ai/schema", + "config_store": { + "enabled": true, + "type": "sqlite", + "config": { + "path": "../../examples/configs/withclickhouselogstorehttp/config.db" + } + }, + "logs_store": { + "enabled": true, + "type": "clickhouse", + "retention_days": 30, + "config": { + "host": "localhost", + "port": "8123", + "database": "bifrost", + "username": "bifrost", + "password": "bifrost_password", + "protocol": "http" + } + }, + "providers": { + "openai": { + "keys": [ + { + "name": "openai-key-1", + "value": "sk-proj-abc", + "weight": 1, + "models": ["*"] + } + ] + } + } +} diff --git a/examples/mcps/http-no-ping-server/main.go b/examples/mcps/http-no-ping-server/main.go index 8364e6a0cb9..e0201a69646 100644 --- a/examples/mcps/http-no-ping-server/main.go +++ b/examples/mcps/http-no-ping-server/main.go @@ -7,6 +7,7 @@ import ( "io" "log" "net/http" + "os" "strings" "github.com/mark3labs/mcp-go/mcp" @@ -51,8 +52,13 @@ func main() { // Create HTTP server using StreamableHTTP transport httpServer := server.NewStreamableHTTPServer(mcpServer) - port := 3001 - addr := fmt.Sprintf("localhost:%d", port) + // Port defaults to 3001 but can be overridden via MCP_SERVER_PORT so multiple + // instances (or a test harness facing a port conflict) can bind elsewhere. + port := "3001" + if p := os.Getenv("MCP_SERVER_PORT"); p != "" { + port = p + } + addr := fmt.Sprintf("localhost:%s", port) log.Printf("MCP server listening on http://%s/", addr) log.Printf("Note: This server does NOT support ping. Use is_ping_available=false in Bifrost config.") diff --git a/examples/mcps/mcp-test-client/README.md b/examples/mcps/mcp-test-client/README.md new file mode 100644 index 00000000000..bb887098b7b --- /dev/null +++ b/examples/mcps/mcp-test-client/README.md @@ -0,0 +1,80 @@ +# mcp-test-client + +A tiny interactive MCP client for exercising a running Bifrost `/mcp` endpoint +under any inbound-auth configuration. It speaks streamable HTTP via +[`mark3labs/mcp-go`](https://github.com/mark3labs/mcp-go) and supports both +credential styles the Bifrost MCP server accepts: + +- **header credentials** — a virtual key (`x-bf-vk`, `Authorization: Bearer`, or + `x-api-key`) or a session id (`x-bf-mcp-session-id`). Use with server auth + mode `headers` or `both`. +- **OAuth** — full discovery (RFC 9728/8414) + dynamic client registration + + PKCE authorization-code flow, with a local browser-callback. Use with server + auth mode `both` or `oauth`. + +Toggle the server-side knobs on your running instance (`mcp_server_auth_mode`, +`enforce_auth_on_inference`, `disable_vk_identity`, virtual-key active state, +...), then `reconnect` and `list` / `call` tools to see the effect. The OAuth +token is held in memory for the session, so `reconnect` after a knob flip does +not re-prompt unless the token is actually rejected. + +## Build / run + +```bash +cd examples/mcps/mcp-test-client +GOWORK=off go run . [flags] +``` + +## Examples + +Virtual key over `x-bf-vk` (headers / both mode): + +```bash +GOWORK=off go run . -url http://localhost:8080/mcp -auth headers -vk sk-bf-xxxxx +``` + +Same VK as a bearer, or as `x-api-key`: + +```bash +GOWORK=off go run . -auth headers -bearer sk-bf-xxxxx +GOWORK=off go run . -auth headers -api-key sk-bf-xxxxx +``` + +Session id (only accepted while `enforce_auth_on_inference=false`): + +```bash +GOWORK=off go run . -auth headers -session +``` + +OAuth (both / oauth mode) — opens a browser for consent, registers dynamically: + +```bash +GOWORK=off go run . -auth oauth -scope mcp +``` + +Anonymous (no creds), or one-shot non-interactive: + +```bash +GOWORK=off go run . -auth headers # anonymous +GOWORK=off go run . -auth headers -vk sk-bf-xxx -once list +GOWORK=off go run . -auth headers -vk sk-bf-xxx -once 'call echo {"text":"hi"}' +``` + +## REPL commands + +``` +list | tools list tools visible to the current credential +desc show a tool's description + input schema +call [json] call a tool, e.g. call echo {"text":"hi"} +set
change/add a header (then 'reconnect' to apply) +unset
remove a header (then 'reconnect' to apply) +reconnect redo start+initialize (use after toggling server knobs) +info show current url / auth / headers +help this text +quit | exit leave +``` + +## Notes + +- This module is intentionally outside the repo `go.work`; run it with + `GOWORK=off` (or build a binary) so it resolves against its own `go.mod`. \ No newline at end of file diff --git a/examples/mcps/mcp-test-client/go.mod b/examples/mcps/mcp-test-client/go.mod new file mode 100644 index 00000000000..99882b1e844 --- /dev/null +++ b/examples/mcps/mcp-test-client/go.mod @@ -0,0 +1,17 @@ +module mcp-test-client + +go 1.26.4 + +require github.com/mark3labs/mcp-go v0.43.2 + +require ( + github.com/bahlo/generic-list-go v0.2.0 // indirect + github.com/buger/jsonparser v1.1.2 // indirect + github.com/google/uuid v1.6.0 // indirect + github.com/invopop/jsonschema v0.13.0 // indirect + github.com/mailru/easyjson v0.7.7 // indirect + github.com/spf13/cast v1.7.1 // indirect + github.com/wk8/go-ordered-map/v2 v2.1.8 // indirect + github.com/yosida95/uritemplate/v3 v3.0.2 // indirect + gopkg.in/yaml.v3 v3.0.1 // indirect +) diff --git a/examples/mcps/mcp-test-client/go.sum b/examples/mcps/mcp-test-client/go.sum new file mode 100644 index 00000000000..a3ebc452ddc --- /dev/null +++ b/examples/mcps/mcp-test-client/go.sum @@ -0,0 +1,39 @@ +github.com/bahlo/generic-list-go v0.2.0 h1:5sz/EEAK+ls5wF+NeqDpk5+iNdMDXrh3z3nPnH1Wvgk= +github.com/bahlo/generic-list-go v0.2.0/go.mod h1:2KvAjgMlE5NNynlg/5iLrrCCZ2+5xWbdbCW3pNTGyYg= +github.com/buger/jsonparser v1.1.2 h1:frqHqw7otoVbk5M8LlE/L7HTnIq2v9RX6EJ48i9AxJk= +github.com/buger/jsonparser v1.1.2/go.mod h1:6RYKKt7H4d4+iWqouImQ9R2FZql3VbhNgx27UK13J/0= +github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c= +github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= +github.com/frankban/quicktest v1.14.6 h1:7Xjx+VpznH+oBnejlPUj8oUpdxnVs4f8XU8WnHkI4W8= +github.com/frankban/quicktest v1.14.6/go.mod h1:4ptaffx2x8+WTWXmUCuVU6aPUX1/Mz7zb5vbUoiM6w0= +github.com/google/go-cmp v0.5.9 h1:O2Tfq5qg4qc4AmwVlvv0oLiVAGB7enBSJ2x2DqQFi38= +github.com/google/go-cmp v0.5.9/go.mod h1:17dUlkBOakJ0+DkrSSNjCkIjxS6bF9zb3elmeNGIjoY= +github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= +github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo= +github.com/invopop/jsonschema v0.13.0 h1:KvpoAJWEjR3uD9Kbm2HWJmqsEaHt8lBUpd0qHcIi21E= +github.com/invopop/jsonschema v0.13.0/go.mod h1:ffZ5Km5SWWRAIN6wbDXItl95euhFz2uON45H2qjYt+0= +github.com/josharian/intern v1.0.0/go.mod h1:5DoeVV0s6jJacbCEi61lwdGj/aVlrQvzHFFd8Hwg//Y= +github.com/kr/pretty v0.3.1 h1:flRD4NNwYAUpkphVc1HcthR4KEIFJ65n8Mw5qdRn3LE= +github.com/kr/pretty v0.3.1/go.mod h1:hoEshYVHaxMs3cyo3Yncou5ZscifuDolrwPKZanG3xk= +github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY= +github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE= +github.com/mailru/easyjson v0.7.7 h1:UGYAvKxe3sBsEDzO8ZeWOSlIQfWFlxbzLZe7hwFURr0= +github.com/mailru/easyjson v0.7.7/go.mod h1:xzfreul335JAWq5oZzymOObrkdz5UnU4kGfJJLY9Nlc= +github.com/mark3labs/mcp-go v0.43.2 h1:21PUSlWWiSbUPQwXIJ5WKlETixpFpq+WBpbMGDSVy/I= +github.com/mark3labs/mcp-go v0.43.2/go.mod h1:YnJfOL382MIWDx1kMY+2zsRHU/q78dBg9aFb8W6Thdw= +github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZbAQM= +github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4= +github.com/rogpeppe/go-internal v1.9.0 h1:73kH8U+JUqXU8lRuOHeVHaa/SZPifC7BkcraZVejAe8= +github.com/rogpeppe/go-internal v1.9.0/go.mod h1:WtVeX8xhTBvf0smdhujwtBcq4Qrzq/fJaraNFVN+nFs= +github.com/spf13/cast v1.7.1 h1:cuNEagBQEHWN1FnbGEjCXL2szYEXqfJPbP2HNUaca9Y= +github.com/spf13/cast v1.7.1/go.mod h1:ancEpBxwJDODSW/UG4rDrAqiKolqNNh2DX3mk86cAdo= +github.com/stretchr/testify v1.9.0 h1:HtqpIVDClZ4nwg75+f6Lvsy/wHu+3BoSGCbBAcpTsTg= +github.com/stretchr/testify v1.9.0/go.mod h1:r2ic/lqez/lEtzL7wO/rwa5dbSLXVDPFyf8C91i36aY= +github.com/wk8/go-ordered-map/v2 v2.1.8 h1:5h/BUHu93oj4gIdvHHHGsScSTMijfx5PeYkE/fJgbpc= +github.com/wk8/go-ordered-map/v2 v2.1.8/go.mod h1:5nJHM5DyteebpVlHnWMV0rPz6Zp7+xBAnxjb1X5vnTw= +github.com/yosida95/uritemplate/v3 v3.0.2 h1:Ed3Oyj9yrmi9087+NczuL5BwkIc4wvTb5zIM+UJPGz4= +github.com/yosida95/uritemplate/v3 v3.0.2/go.mod h1:ILOh0sOhIJR3+L/8afwt/kE++YT040gmv5BQTMR2HP4= +gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM= +gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= +gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= +gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= diff --git a/examples/mcps/mcp-test-client/main.go b/examples/mcps/mcp-test-client/main.go new file mode 100644 index 00000000000..6cf545a5877 --- /dev/null +++ b/examples/mcps/mcp-test-client/main.go @@ -0,0 +1,516 @@ +// Command mcp-test-client is a small interactive MCP client for exercising a +// running Bifrost /mcp endpoint under any inbound-auth configuration. +// +// It speaks streamable HTTP and supports the two credential styles Bifrost's +// MCP server accepts: +// +// - header credentials: a virtual key (x-bf-vk / Authorization: Bearer / +// x-api-key) or a session id (x-bf-mcp-session-id), used for the +// `headers` and `both` server auth modes; and +// - OAuth: full RFC 9728/8414 discovery + dynamic client registration + PKCE +// authorization-code flow, used for the `both` and `oauth` server modes. +// +// Toggle the server-side knobs (mcp_server_auth_mode, enforce_auth_on_inference, +// disable_vk_identity, virtual-key state, ...) however you like on the running +// instance, then `reconnect` here and `list` / `call` tools to observe the +// effect. The OAuth token is held in memory for the session, so `reconnect` +// after a knob flip does not re-prompt unless the token is actually rejected. +package main + +import ( + "bufio" + "context" + "encoding/json" + "flag" + "fmt" + "net" + "net/http" + "os" + "os/exec" + "runtime" + "strings" + "time" + + "github.com/mark3labs/mcp-go/client" + "github.com/mark3labs/mcp-go/client/transport" + "github.com/mark3labs/mcp-go/mcp" +) + +// oauthHTTPTimeout bounds each outbound OAuth request (discovery, dynamic +// registration, authorization-URL build, token exchange) so a stalled IdP +// surfaces a recoverable error instead of wedging the CLI. +const oauthHTTPTimeout = 30 * time.Second + +type config struct { + url string + auth string // "headers" or "oauth" + headers map[string]string + scopes []string + callbackPort string + clientID string + tokenStore client.TokenStore // persisted across reconnects so OAuth is not re-run needlessly +} + +// headerList collects repeated -header "Key: Value" flags. +type headerList map[string]string + +func (h headerList) String() string { return fmt.Sprintf("%v", map[string]string(h)) } +func (h headerList) Set(v string) error { + k, val, ok := strings.Cut(v, ":") + if !ok { + return fmt.Errorf("expected \"Key: Value\", got %q", v) + } + k = strings.TrimSpace(k) + if k == "" { + return fmt.Errorf("expected non-empty header name in %q", v) + } + h[k] = strings.TrimSpace(val) + return nil +} + +func main() { + extra := headerList{} + url := flag.String("url", "http://localhost:8080/mcp", "Bifrost /mcp endpoint") + auth := flag.String("auth", "headers", "credential style: headers | oauth") + vk := flag.String("vk", "", "virtual key, sent as x-bf-vk (auth=headers)") + bearer := flag.String("bearer", "", "virtual key sent as Authorization: Bearer (auth=headers)") + apiKey := flag.String("api-key", "", "virtual key sent as x-api-key (auth=headers)") + session := flag.String("session", "", "session id, sent as x-bf-mcp-session-id (auth=headers)") + flag.Var(extra, "header", "extra header \"Key: Value\" (repeatable, any auth)") + scope := flag.String("scope", "mcp", "comma-separated OAuth scopes (auth=oauth)") + cbPort := flag.String("callback-port", "8585", "local port for the OAuth redirect callback (auth=oauth)") + clientID := flag.String("client-id", "", "pre-registered OAuth client id; empty uses dynamic registration (auth=oauth)") + once := flag.String("once", "", "run a single command and exit, e.g. -once list or -once 'call echo {\"text\":\"hi\"}'") + flag.Parse() + + headers := map[string]string{} + if *vk != "" { + headers["x-bf-vk"] = *vk + } + if *bearer != "" { + headers["Authorization"] = "Bearer " + *bearer + } + if *apiKey != "" { + headers["x-api-key"] = *apiKey + } + if *session != "" { + headers["x-bf-mcp-session-id"] = *session + } + for k, v := range extra { + headers[k] = v + } + + cfg := &config{ + url: *url, + auth: strings.ToLower(*auth), + headers: headers, + scopes: splitNonEmpty(*scope), + callbackPort: *cbPort, + clientID: *clientID, + tokenStore: client.NewMemoryTokenStore(), + } + + c, err := connect(cfg) + if err != nil { + fmt.Fprintf(os.Stderr, "connect failed: %v\n", err) + os.Exit(1) + } + defer c.Close() + + if *once != "" { + if err := dispatch(c, *once); err != nil { + fmt.Fprintf(os.Stderr, "%v\n", err) + os.Exit(1) + } + return + } + + printInfo(cfg) + fmt.Println("Type 'help' for commands.") + repl(cfg, c) +} + +// connect builds a client for the current config, starts the transport and runs +// the initialize handshake, driving the OAuth browser flow on demand. +func connect(cfg *config) (*client.Client, error) { + var c *client.Client + var err error + switch cfg.auth { + case "oauth": + oauthCfg := client.OAuthConfig{ + ClientID: cfg.clientID, + RedirectURI: fmt.Sprintf("http://localhost:%s/oauth/callback", cfg.callbackPort), + Scopes: cfg.scopes, + TokenStore: cfg.tokenStore, + PKCEEnabled: true, + // Dedicated client (not http.DefaultClient) so a stalled IdP during + // discovery / registration / token exchange surfaces as an error + // instead of wedging the CLI. + HTTPClient: &http.Client{Timeout: oauthHTTPTimeout}, + } + c, err = client.NewOAuthStreamableHttpClient(cfg.url, oauthCfg, transport.WithHTTPHeaders(cfg.headers)) + case "headers", "": + c, err = client.NewStreamableHttpClient(cfg.url, transport.WithHTTPHeaders(cfg.headers)) + default: + return nil, fmt.Errorf("unknown -auth %q (want headers | oauth)", cfg.auth) + } + if err != nil { + return nil, err + } + + ctx := context.Background() + if err := withAuthRetry(cfg, c.Start(ctx), func() error { return c.Start(ctx) }); err != nil { + return nil, fmt.Errorf("start: %w", err) + } + + res, err := c.Initialize(ctx, initRequest()) + if err != nil { + if err = withAuthRetry(cfg, err, func() error { + var e error + res, e = c.Initialize(ctx, initRequest()) + return e + }); err != nil { + return nil, fmt.Errorf("initialize: %w", err) + } + } + fmt.Printf("connected to %s %s\n", res.ServerInfo.Name, res.ServerInfo.Version) + return c, nil +} + +// withAuthRetry runs the OAuth authorization flow if err signals it, then calls +// retry once. For non-OAuth errors it returns err unchanged. +func withAuthRetry(cfg *config, err error, retry func() error) error { + if err == nil { + return nil + } + if !client.IsOAuthAuthorizationRequiredError(err) { + return err + } + if aerr := authorize(cfg, client.GetOAuthHandler(err)); aerr != nil { + return aerr + } + return retry() +} + +func initRequest() mcp.InitializeRequest { + var req mcp.InitializeRequest + req.Params.ProtocolVersion = mcp.LATEST_PROTOCOL_VERSION + req.Params.ClientInfo = mcp.Implementation{Name: "bifrost-mcp-test-client", Version: "0.1.0"} + req.Params.Capabilities = mcp.ClientCapabilities{} + return req +} + +func repl(cfg *config, c *client.Client) { + sc := bufio.NewScanner(os.Stdin) + sc.Buffer(make([]byte, 0, 64*1024), 1<<20) + for { + fmt.Print("mcp> ") + if !sc.Scan() { + return + } + line := strings.TrimSpace(sc.Text()) + if line == "" { + continue + } + switch firstWord(line) { + case "quit", "exit": + return + case "help": + printHelp() + case "info": + printInfo(cfg) + case "reconnect": + c.Close() + nc, err := connect(cfg) + if err != nil { + fmt.Fprintf(os.Stderr, "reconnect failed: %v\n", err) + continue + } + c = nc + case "set": + if err := setHeader(cfg, line); err != nil { + fmt.Fprintf(os.Stderr, "%v\n", err) + } + case "unset": + delete(cfg.headers, strings.TrimSpace(rest(line))) + fmt.Println("(run 'reconnect' to apply)") + default: + if err := dispatch(c, line); err != nil { + fmt.Fprintf(os.Stderr, "%v\n", err) + } + } + } +} + +// dispatch runs a single tool command (list / call / desc). +func dispatch(c *client.Client, line string) error { + ctx := context.Background() + switch firstWord(line) { + case "list", "tools": + res, err := c.ListTools(ctx, mcp.ListToolsRequest{}) + if err != nil { + return err + } + if len(res.Tools) == 0 { + fmt.Println("(no tools exposed for this credential)") + } + for _, t := range res.Tools { + fmt.Printf("- %s\t%s\n", t.Name, firstLine(t.Description)) + } + return nil + case "desc": + name := strings.TrimSpace(rest(line)) + res, err := c.ListTools(ctx, mcp.ListToolsRequest{}) + if err != nil { + return err + } + for _, t := range res.Tools { + if t.Name == name { + b, _ := json.MarshalIndent(t.InputSchema, "", " ") + fmt.Printf("%s\n%s\n", t.Description, b) + return nil + } + } + return fmt.Errorf("tool %q not found", name) + case "call": + name, argStr := splitFirst(rest(line)) + if name == "" { + return fmt.Errorf("usage: call [json-args]") + } + var args any + if s := strings.TrimSpace(argStr); s != "" { + if err := json.Unmarshal([]byte(s), &args); err != nil { + return fmt.Errorf("invalid JSON args: %w", err) + } + } + var req mcp.CallToolRequest + req.Params.Name = name + req.Params.Arguments = args + res, err := c.CallTool(ctx, req) + if err != nil { + return err + } + printToolResult(res) + return nil + default: + return fmt.Errorf("unknown command %q (try 'help')", firstWord(line)) + } +} + +func printToolResult(res *mcp.CallToolResult) { + if res.IsError { + fmt.Println("[tool returned isError=true]") + } + for _, content := range res.Content { + if tc, ok := mcp.AsTextContent(content); ok { + fmt.Println(tc.Text) + } else { + b, _ := json.Marshal(content) + fmt.Println(string(b)) + } + } + if res.StructuredContent != nil { + b, _ := json.MarshalIndent(res.StructuredContent, "", " ") + fmt.Printf("structuredContent: %s\n", b) + } +} + +func setHeader(cfg *config, line string) error { + kv := strings.TrimSpace(rest(line)) + // Split on whichever separator appears first — colon or space — so the + // "Key: Value" colon form (matching the -header flag syntax) is stored as + // key "Authorization", not "Authorization:". Header names contain neither a + // colon nor a space, so the first one is always the name/value boundary. + i := strings.IndexAny(kv, ": \t") + if i < 0 { + return fmt.Errorf("usage: set ") + } + k := strings.TrimSpace(kv[:i]) + v := strings.TrimSpace(kv[i+1:]) + if k == "" { + return fmt.Errorf("usage: set ") + } + cfg.headers[k] = v + fmt.Println("(run 'reconnect' to apply)") + return nil +} + +// authorize runs the interactive OAuth authorization-code flow: register a +// client if needed, open the browser, capture the callback, exchange the code. +func authorize(cfg *config, h *transport.OAuthHandler) error { + fmt.Println("OAuth authorization required — starting the flow...") + + codeVerifier, err := client.GenerateCodeVerifier() + if err != nil { + return err + } + codeChallenge := client.GenerateCodeChallenge(codeVerifier) + state, err := client.GenerateState() + if err != nil { + return err + } + + if h.GetClientID() == "" { + regCtx, cancel := context.WithTimeout(context.Background(), oauthHTTPTimeout) + err := h.RegisterClient(regCtx, "bifrost-mcp-test-client") + cancel() + if err != nil { + return fmt.Errorf("dynamic client registration: %w", err) + } + } + + urlCtx, cancel := context.WithTimeout(context.Background(), oauthHTTPTimeout) + authURL, err := h.GetAuthorizationURL(urlCtx, state, codeChallenge) + cancel() + if err != nil { + return err + } + + cbChan := make(chan map[string]string, 1) + mux := http.NewServeMux() + mux.HandleFunc("/oauth/callback", func(w http.ResponseWriter, r *http.Request) { + params := map[string]string{} + for k, vs := range r.URL.Query() { + if len(vs) > 0 { + params[k] = vs[0] + } + } + w.Header().Set("Content-Type", "text/html") + _, _ = w.Write([]byte("

Authorization received

You can close this tab.")) + cbChan <- params + }) + srv := &http.Server{ + Handler: mux, + ReadHeaderTimeout: 5 * time.Second, + ReadTimeout: 10 * time.Second, + WriteTimeout: 10 * time.Second, + IdleTimeout: 30 * time.Second, + } + ln, err := net.Listen("tcp", "localhost:"+cfg.callbackPort) + if err != nil { + return fmt.Errorf("listen on callback port %s: %w", cfg.callbackPort, err) + } + defer srv.Close() + go srv.Serve(ln) + + fmt.Printf("Open this URL to authorize:\n %s\n", authURL) + openBrowser(authURL) + + // Bound the wait so a callback that never arrives (browser didn't open, tab + // closed, or the AS errored before redirecting) surfaces a recoverable error + // and lets the deferred srv.Close run, instead of hanging until the user + // kills the process. + var params map[string]string + select { + case params = <-cbChan: + case <-time.After(3 * time.Minute): + return fmt.Errorf("timed out waiting for the OAuth callback; re-run the command to retry") + } + if e := params["error"]; e != "" { + return fmt.Errorf("authorization error: %s %s", e, params["error_description"]) + } + if params["state"] != state { + return fmt.Errorf("state mismatch (possible CSRF)") + } + code := params["code"] + if code == "" { + return fmt.Errorf("no authorization code in callback") + } + exCtx, cancel := context.WithTimeout(context.Background(), oauthHTTPTimeout) + err = h.ProcessAuthorizationResponse(exCtx, code, state, codeVerifier) + cancel() + if err != nil { + return fmt.Errorf("code exchange: %w", err) + } + fmt.Println("authorization complete.") + return nil +} + +func openBrowser(url string) { + var cmd string + var args []string + switch runtime.GOOS { + case "darwin": + cmd = "open" + case "windows": + cmd, args = "rundll32", []string{"url.dll,FileProtocolHandler"} + default: + cmd = "xdg-open" + } + _ = exec.Command(cmd, append(args, url)...).Start() +} + +func printInfo(cfg *config) { + fmt.Printf("url: %s\nauth: %s\n", cfg.url, cfg.auth) + if cfg.auth == "oauth" { + fmt.Printf("scopes: %v callback: http://localhost:%s/oauth/callback\n", cfg.scopes, cfg.callbackPort) + } + if len(cfg.headers) > 0 { + fmt.Println("headers:") + for k, v := range cfg.headers { + fmt.Printf(" %s: %s\n", k, mask(v)) + } + } +} + +func printHelp() { + fmt.Print(`commands: + list | tools list tools visible to the current credential + desc show a tool's description + input schema + call [json] call a tool, e.g. call echo {"text":"hi"} + set
change/add a header (then 'reconnect' to apply) + unset
remove a header (then 'reconnect' to apply) + reconnect redo start+initialize (use after toggling server knobs) + info show current url / auth / headers + help this text + quit | exit leave +`) +} + +// helpers + +func splitNonEmpty(s string) []string { + var out []string + for _, p := range strings.Split(s, ",") { + if p = strings.TrimSpace(p); p != "" { + out = append(out, p) + } + } + return out +} + +func firstWord(s string) string { + w, _ := splitFirst(s) + return w +} + +func rest(s string) string { + _, r := splitFirst(s) + return r +} + +func splitFirst(s string) (string, string) { + s = strings.TrimSpace(s) + if i := strings.IndexAny(s, " \t"); i >= 0 { + return s[:i], strings.TrimSpace(s[i+1:]) + } + return s, "" +} + +func firstLine(s string) string { + if i := strings.IndexByte(s, '\n'); i >= 0 { + return s[:i] + } + return s +} + +// mask redacts a header value for display. Every header this client sends is a +// credential or test input the user typed, so all values are masked rather than +// matching a substring allowlist (which would leak custom auth headers like +// Cookie or X-Token); the prefix/suffix still lets the user confirm the value. +func mask(val string) string { + if len(val) <= 8 { + return "****" + } + return val[:4] + "..." + val[len(val)-4:] +} diff --git a/examples/plugins/hello-world/.gitignore b/examples/plugins/hello-world/.gitignore deleted file mode 100644 index 76de1335085..00000000000 --- a/examples/plugins/hello-world/.gitignore +++ /dev/null @@ -1,15 +0,0 @@ -# Build artifacts -build/ -*.so -*.dll -*.dylib - -# Go build cache -*.exe -*.exe~ -*.test -*.out - -# Dependency directories -vendor/ - diff --git a/framework/changelog.md b/framework/changelog.md index e69de29bb2d..0f610f2bbd3 100644 --- a/framework/changelog.md +++ b/framework/changelog.md @@ -0,0 +1,25 @@ +- feat: added ClickHouse support for the log store with a hybrid store mode (#4748, #4893) +- feat: added OAuth 2.1 gateway auth: AS discovery endpoints, signing key management, `MCPServerAuthMode` config, issuance endpoints (DCR, authorize, token) with PKCE and refresh token rotation, session listing, revocation, family-revocation, VK liveness checks, and sweep worker (#4505, #4506, #4509) +- feat: revoke VK-mode OAuth2 grants on VK deletion with user-liveness checks at refresh and request time (#4806) +- feat: push OAuth2 sessions filtering and pagination to SQL with total count (#4775) +- feat: added expiry field to virtual keys (#4887) +- feat: virtual key values use `schemas.SecretVar` to support the env store (#4817) +- feat: added `BedrockMantleKeyConfig` support to key hashing, schema/table mapping, and sensitive field clearing (#4737, #4886) +- feat: added per-MCP-server tool execution timeout (#4472, closes #4446) (thanks [@Purvi09](https://github.com/Purvi09)!) +- feat: added connection_type, auth_type, state, virtual_key, and server/client_id filters with pagination to the MCP clients list (#4839, #4767, #4841) +- feat: added `user_name`, `team_ids`, `team_names`, `customer_ids`, `customer_names`, `business_unit_ids`, and `business_unit_names` to log list select columns (#4866) +- feat: added `UpsertModelParametersBatch` for batched model parameter sync (#4800) +- feat: added `is_deprecated` to pricing and catalog responses and mark deprecated models instead of filtering them (#4779, #4792, #4936) +- feat: drop reasoning when tools are present but `reasoning_with_tool_calls` is unsupported (#4630) +- feat: extended vendor-prefix pricing fallback to OpenAI, Google, and xAI models (#4924) +- feat: lowered `auth_code_ttl` default to 300s and enforce a 900s maximum (#4822) +- fix: sweep orphaned deferred spans in trace store TTL cleanup (#4869, closes #4868) (thanks [@citrocat](https://github.com/citrocat)!) +- fix: rebuild token usage from denormalized columns in hybrid log list (#4722, closes #4721) (thanks [@G-XD](https://github.com/G-XD)!) +- fix: stats for cancelled requests (#4930) +- fix: tier costs evaluated via input tokens instead of total tokens (#4917) +- fix: append datasheet models for incomplete list-models calls (#4879) +- fix: skip background token refresh for disabled or unconfigured MCP clients and guarantee non-nil logger in sync workers (#4848) +- fix: exclude terminal-status OAuth configs from the expiring token refresh query (#4754) +- fix: check whether virtual key values are secrets (#4927) +- fix: web fetch fixes (#4945) +- fix: Perplexity Responses API compatibility (#4813) diff --git a/framework/configstore/clientconfig.go b/framework/configstore/clientconfig.go index 72a95a32265..31766a3f32a 100644 --- a/framework/configstore/clientconfig.go +++ b/framework/configstore/clientconfig.go @@ -10,6 +10,7 @@ import ( "math" "sort" "strconv" + "time" "github.com/bytedance/sonic" bifrost "github.com/maximhq/bifrost/core" @@ -98,10 +99,19 @@ type ClientConfig struct { HideDeletedVirtualKeysInFilters bool `json:"hide_deleted_virtual_keys_in_filters"` // Hide deleted virtual keys from logs/MCP filter data RoutingChainMaxDepth int `json:"routing_chain_max_depth"` // Maximum depth for routing rule chain evaluation (default: 10) MCPExternalClientURL *schemas.SecretVar `json:"mcp_external_client_url,omitempty"` // Public base URL used as redirect_uri when Bifrost acts as an OAuth client to upstream MCP servers. Supports env var syntax ("env.MY_VAR") + MCPServerAuthMode tables.MCPServerAuthMode `json:"mcp_server_auth_mode,omitempty"` // How /mcp authenticates inbound clients: headers (default), both, or oauth. + OAuth2ServerConfig *tables.OAuth2ServerConfig `json:"oauth2_server_config,omitempty"` // OAuth2 AS-specific settings (IssuerURL, token TTLs). Only relevant when MCPServerAuthMode is both or oauth. ConfigHash string `json:"-"` // Config hash for reconciliation (not serialized) DumpErrorsInConsoleLogs bool `json:"dump_errors_in_console_logs"` // Dump error details in console logs } +// IsMCPOAuthDiscoveryEnabled reports whether the well-known OAuth discovery +// endpoints and JWKS endpoint should be live. True when MCPServerAuthMode is +// both or oauth. +func (c *ClientConfig) IsMCPOAuthDiscoveryEnabled() bool { + return c.MCPServerAuthMode == tables.MCPServerAuthModeBoth || c.MCPServerAuthMode == tables.MCPServerAuthModeOAuth +} + // UnmarshalJSON defaults all bool fields to true when absent from JSON. func (c *ClientConfig) UnmarshalJSON(data []byte) error { type ClientConfigAlias ClientConfig @@ -372,6 +382,26 @@ func (c *ClientConfig) GenerateClientConfigHash() (string, error) { } } + // Only hash non-default values to avoid legacy config hash churn on upgrade — + // existing configs carry an empty auth mode and a nil OAuth2 server config. + if c.MCPServerAuthMode != "" { + hash.Write([]byte("mcpServerAuthMode:" + string(c.MCPServerAuthMode))) + } + // Hash OAuth2ServerConfig field-by-field (not via Marshal) for a stable, + // deterministic byte stream that does not depend on serializer field order. + if c.OAuth2ServerConfig != nil { + oc := c.OAuth2ServerConfig + if oc.IssuerURL.IsSet() { + if oc.IssuerURL.IsFromEnv() { + hash.Write([]byte("oauth2IssuerURL:env:" + oc.IssuerURL.GetRawRef())) + } else { + hash.Write([]byte("oauth2IssuerURL:val:" + oc.IssuerURL.GetValue())) + } + } + hash.Write([]byte("oauth2AuthCodeTTL:" + strconv.Itoa(oc.AuthCodeTTL))) + hash.Write([]byte("oauth2AccessTokenTTL:" + strconv.Itoa(oc.AccessTokenTTL))) + } + return hex.EncodeToString(hash.Sum(nil)), nil } @@ -547,6 +577,29 @@ func (p *ProviderConfig) Redacted() *ProviderConfig { redactedConfig.Keys[i].BedrockKeyConfig = bedrockConfig } + // Redact Bedrock Mantle key config if present + if key.BedrockMantleKeyConfig != nil { + mantleConfig := &schemas.BedrockMantleKeyConfig{} + mantleConfig.AccessKey = *key.BedrockMantleKeyConfig.AccessKey.Redacted() + mantleConfig.SecretKey = *key.BedrockMantleKeyConfig.SecretKey.Redacted() + if key.BedrockMantleKeyConfig.SessionToken != nil { + mantleConfig.SessionToken = key.BedrockMantleKeyConfig.SessionToken.Redacted() + } + if key.BedrockMantleKeyConfig.Region != nil { + mantleConfig.Region = key.BedrockMantleKeyConfig.Region.Redacted() + } + if key.BedrockMantleKeyConfig.RoleARN != nil { + mantleConfig.RoleARN = key.BedrockMantleKeyConfig.RoleARN.Redacted() + } + if key.BedrockMantleKeyConfig.ExternalID != nil { + mantleConfig.ExternalID = key.BedrockMantleKeyConfig.ExternalID.Redacted() + } + if key.BedrockMantleKeyConfig.RoleSessionName != nil { + mantleConfig.RoleSessionName = key.BedrockMantleKeyConfig.RoleSessionName.Redacted() + } + redactedConfig.Keys[i].BedrockMantleKeyConfig = mantleConfig + } + if key.VLLMKeyConfig != nil { vllmConfig := &schemas.VLLMKeyConfig{ ModelName: key.VLLMKeyConfig.ModelName, @@ -715,6 +768,14 @@ func GenerateKeyHash(key schemas.Key) (string, error) { } hash.Write(data) } + // Hash BedrockMantleKeyConfig + if key.BedrockMantleKeyConfig != nil { + data, err := sonic.Marshal(key.BedrockMantleKeyConfig) + if err != nil { + return "", err + } + hash.Write(data) + } // Hash Aliases if key.Aliases != nil { data, err := sonic.Marshal(key.Aliases) @@ -811,14 +872,19 @@ func GenerateVirtualKeyHash(vk tables.TableVirtualKey) (string, error) { hash.Write([]byte(vk.Name)) // Hash Description hash.Write([]byte(vk.Description)) - // Hash Value - hash.Write([]byte(vk.Value)) + // Hash the resolved value so that secret rotation (vault/env change) is + // detected as a config change and triggers a re-sync. + hash.Write([]byte(vk.Value.GetValue())) // Hash IsActive (nil treated as DB default true) if vk.IsActiveValue() { hash.Write([]byte("isActive:true")) } else { hash.Write([]byte("isActive:false")) } + // Hash ExpiresAt only when set, so rows created before expiry existed keep their hash + if vk.ExpiresAt != nil { + hash.Write([]byte("expiresAt:" + vk.ExpiresAt.UTC().Format(time.RFC3339Nano))) + } // Hash TeamID if vk.TeamID != nil { hash.Write([]byte("teamID:" + *vk.TeamID)) diff --git a/framework/configstore/encryption_test.go b/framework/configstore/encryption_test.go index 10420085f10..60233f88b5a 100644 --- a/framework/configstore/encryption_test.go +++ b/framework/configstore/encryption_test.go @@ -428,7 +428,7 @@ func TestEncryptPlaintextVirtualKeys_EncryptsAndDecryptsCorrectly(t *testing.T) // GORM hooks should decrypt on read var found tables.TableVirtualKey require.NoError(t, db.Where("id = ?", "vk-batch-1").First(&found).Error) - assert.Equal(t, "vk-batch-secret", found.Value) + assert.Equal(t, "vk-batch-secret", found.Value.GetValue()) } func TestEncryptPlaintextOAuthConfigs_EncryptsAndDecryptsCorrectly(t *testing.T) { @@ -1342,7 +1342,7 @@ func TestEncryptPlaintextRows_SkipsAlreadyEncryptedVirtualKeys(t *testing.T) { vk := &tables.TableVirtualKey{ ID: "vk-already-enc", Name: "already-encrypted-vk", - Value: "vk-secret-already", + Value: *schemas.NewSecretVar("vk-secret-already"), IsActive: bifrost.Ptr(true), } require.NoError(t, db.Create(vk).Error) diff --git a/framework/configstore/migrations.go b/framework/configstore/migrations.go index 472bcd2fe10..561b6bdbf33 100644 --- a/framework/configstore/migrations.go +++ b/framework/configstore/migrations.go @@ -429,7 +429,13 @@ var configstoreMigrationSteps = []migrationStep{ {IDs: []string{"add_customer_name_unique_constraint_dedup", "add_customer_name_unique_constraint_index"}, run: migrationAddCustomerNameUniqueConstraint}, {IDs: []string{"null_legacy_customer_budget_id_refs"}, run: migrationNullLegacyCustomerBudgetID}, {IDs: []string{"add_skills_repo_tables"}, run: migrationAddSkillsRepoTables}, + {IDs: []string{"add_oauth2_server_tables"}, run: migrationAddOAuth2ServerTables}, + {IDs: []string{"add_oauth2_issuance_tables"}, run: migrationAddOAuth2IssuanceTables}, {IDs: []string{"add_dump_errors_in_console_logs_column"}, run: migrationAddDumpErrorsInConsoleLogsColumn}, + {IDs: []string{"add_bedrock_mantle_key_columns"}, run: migrationAddBedrockMantleKeyColumns}, + {IDs: []string{"add_model_pricing_is_deprecated_column"}, run: migrationAddModelPricingIsDeprecatedColumn}, + {IDs: []string{"add_mcp_client_tool_execution_timeout_column"}, run: migrationAddMCPClientToolExecutionTimeoutColumn}, + {IDs: []string{"add_virtual_key_expires_at_column"}, run: migrationAddVirtualKeyExpiresAtColumn}, } // quoteSQLiteIdentifier quotes a SQLite identifier, escaping any double quotes. @@ -1110,6 +1116,48 @@ func migrationAddVirtualKeyProviderConfigTable(ctx context.Context, db *gorm.DB, } // migrationAddAllowedOriginsJSONColumn adds the allowed_origins_json column to the client config table +// migrationAddBedrockMantleKeyColumns adds the bedrock_mantle_* SigV4 credential columns to the +// config_keys table for the standalone bedrock_mantle provider. +func migrationAddBedrockMantleKeyColumns(ctx context.Context, db *gorm.DB, logger schemas.Logger) error { + migrationName := "add_bedrock_mantle_key_columns" + logger.Info("[configstore] starting migration %s", migrationName) + defer logger.Info("[configstore] finished migration %s", migrationName) + cols := []string{ + "bedrock_mantle_access_key", + "bedrock_mantle_secret_key", + "bedrock_mantle_session_token", + "bedrock_mantle_region", + "bedrock_mantle_role_arn", + "bedrock_mantle_external_id", + "bedrock_mantle_role_session_name", + } + m := migrator.New(db, migrator.DefaultOptions, []*migrator.Migration{{ + ID: migrationName, + Migrate: func(tx *gorm.DB) error { + tx = tx.WithContext(ctx) + for _, col := range cols { + if err := addColumnIfNotExists(tx, logger, &tables.TableKey{}, col); err != nil { + return err + } + } + return nil + }, + Rollback: func(tx *gorm.DB) error { + tx = tx.WithContext(ctx) + for _, col := range cols { + if err := dropColumnIfExists(tx, logger, &tables.TableKey{}, col); err != nil { + return err + } + } + return nil + }, + }}) + if err := m.Migrate(); err != nil { + return fmt.Errorf("error while running db migration: %s", err.Error()) + } + return nil +} + func migrationAddAllowedOriginsJSONColumn(ctx context.Context, db *gorm.DB, logger schemas.Logger) error { migrationName := "add_allowed_origins_json_column" logger.Info("[configstore] starting migration %s", migrationName) @@ -9612,6 +9660,36 @@ func migrationAddAdditionalAttributesToPricing(ctx context.Context, db *gorm.DB, return nil } +// migrationAddModelPricingIsDeprecatedColumn adds is_deprecated to +// governance_model_pricing so synced datasheets can mark models that should +// remain listable but are no longer recommended for new use. +func migrationAddModelPricingIsDeprecatedColumn(ctx context.Context, db *gorm.DB, logger schemas.Logger) error { + migrationName := "add_model_pricing_is_deprecated_column" + logger.Info("[configstore] starting migration %s", migrationName) + defer logger.Info("[configstore] finished migration %s", migrationName) + m := migrator.New(db, migrator.DefaultOptions, []*migrator.Migration{{ + ID: migrationName, + Migrate: func(tx *gorm.DB) error { + tx = tx.WithContext(ctx) + if err := addColumnIfNotExists(tx, logger, &tables.TableModelPricing{}, "IsDeprecated"); err != nil { + return fmt.Errorf("failed to add is_deprecated column: %w", err) + } + return nil + }, + Rollback: func(tx *gorm.DB) error { + tx = tx.WithContext(ctx) + if err := dropColumnIfExists(tx, logger, &tables.TableModelPricing{}, "IsDeprecated"); err != nil { + return fmt.Errorf("failed to drop is_deprecated column: %w", err) + } + return nil + }, + }}) + if err := m.Migrate(); err != nil { + return fmt.Errorf("error running add_model_pricing_is_deprecated_column migration: %s", err.Error()) + } + return nil +} + // migrationAddCustomerCalendarAlignedColumn adds calendar_aligned to governance_customers // so customer-level calendar alignment can be persisted. No backfill is needed: the // legacy per-budget/per-rate-limit calendar_aligned columns were dropped by @@ -10095,3 +10173,146 @@ func migrationAddCustomerNameUniqueConstraint(ctx context.Context, db *gorm.DB, }, }) } + +func migrationAddOAuth2ServerTables(ctx context.Context, db *gorm.DB, logger schemas.Logger) error { + migrationName := "add_oauth2_server_tables" + logger.Info("[configstore] starting migration %s", migrationName) + defer logger.Info("[configstore] finished migration %s", migrationName) + m := migrator.New(db, migrator.DefaultOptions, []*migrator.Migration{ + { + ID: migrationName, + Migrate: func(tx *gorm.DB) error { + tx = tx.WithContext(ctx) + mg := tx.Migrator() + if !mg.HasColumn(&tables.TableClientConfig{}, "mcp_server_auth_mode") { + if err := mg.AddColumn(&tables.TableClientConfig{}, "MCPServerAuthMode"); err != nil { + return fmt.Errorf("add mcp_server_auth_mode column: %w", err) + } + } + if !mg.HasColumn(&tables.TableClientConfig{}, "oauth2_server_config_json") { + if err := mg.AddColumn(&tables.TableClientConfig{}, "OAuth2ServerConfigJSON"); err != nil { + return fmt.Errorf("add oauth2_server_config_json column: %w", err) + } + } + return nil + }, + Rollback: func(tx *gorm.DB) error { + tx = tx.WithContext(ctx) + mg := tx.Migrator() + if mg.HasColumn(&tables.TableClientConfig{}, "oauth2_server_config_json") { + if err := mg.DropColumn(&tables.TableClientConfig{}, "OAuth2ServerConfigJSON"); err != nil { + return fmt.Errorf("drop oauth2_server_config_json column: %w", err) + } + } + if mg.HasColumn(&tables.TableClientConfig{}, "mcp_server_auth_mode") { + if err := mg.DropColumn(&tables.TableClientConfig{}, "MCPServerAuthMode"); err != nil { + return fmt.Errorf("drop mcp_server_auth_mode column: %w", err) + } + } + return nil + }, + }, + }) + if err := m.Migrate(); err != nil { + return fmt.Errorf("error while running db migration %s: %w", migrationName, err) + } + return nil +} + +func migrationAddOAuth2IssuanceTables(ctx context.Context, db *gorm.DB, logger schemas.Logger) error { + migrationName := "add_oauth2_issuance_tables" + logger.Info("[configstore] starting migration %s", migrationName) + defer logger.Info("[configstore] finished migration %s", migrationName) + m := migrator.New(db, migrator.DefaultOptions, []*migrator.Migration{{ + ID: migrationName, + Migrate: func(tx *gorm.DB) error { + tx = tx.WithContext(ctx) + mg := tx.Migrator() + if !mg.HasTable(&tables.TableOAuth2Client{}) { + if err := mg.CreateTable(&tables.TableOAuth2Client{}); err != nil { + return fmt.Errorf("create oauth2_clients table: %w", err) + } + } + if !mg.HasTable(&tables.TableOAuth2AuthorizeRequest{}) { + if err := mg.CreateTable(&tables.TableOAuth2AuthorizeRequest{}); err != nil { + return fmt.Errorf("create oauth2_authorize_requests table: %w", err) + } + } + if !mg.HasTable(&tables.TableOAuth2RefreshToken{}) { + if err := mg.CreateTable(&tables.TableOAuth2RefreshToken{}); err != nil { + return fmt.Errorf("create oauth2_refresh_tokens table: %w", err) + } + } + return nil + }, + Rollback: func(tx *gorm.DB) error { + tx = tx.WithContext(ctx) + mg := tx.Migrator() + // Drop in reverse creation order. + if mg.HasTable(&tables.TableOAuth2RefreshToken{}) { + if err := mg.DropTable(&tables.TableOAuth2RefreshToken{}); err != nil { + return fmt.Errorf("drop oauth2_refresh_tokens table: %w", err) + } + } + if mg.HasTable(&tables.TableOAuth2AuthorizeRequest{}) { + if err := mg.DropTable(&tables.TableOAuth2AuthorizeRequest{}); err != nil { + return fmt.Errorf("drop oauth2_authorize_requests table: %w", err) + } + } + if mg.HasTable(&tables.TableOAuth2Client{}) { + if err := mg.DropTable(&tables.TableOAuth2Client{}); err != nil { + return fmt.Errorf("drop oauth2_clients table: %w", err) + } + } + return nil + }, + }}) + if err := m.Migrate(); err != nil { + return fmt.Errorf("error while running db migration %s: %w", migrationName, err) + } + return nil +} + +func migrationAddMCPClientToolExecutionTimeoutColumn(ctx context.Context, db *gorm.DB, logger schemas.Logger) error { + migrationName := "add_mcp_client_tool_execution_timeout_column" + logger.Info("[configstore] starting migration %s", migrationName) + defer logger.Info("[configstore] finished migration %s", migrationName) + m := migrator.New(db, migrator.DefaultOptions, []*migrator.Migration{{ + ID: migrationName, + Migrate: func(tx *gorm.DB) error { + tx = tx.WithContext(ctx) + return addColumnIfNotExists(tx, logger, &tables.TableMCPClient{}, "tool_execution_timeout") + }, + Rollback: func(tx *gorm.DB) error { + tx = tx.WithContext(ctx) + return dropColumnIfExists(tx, logger, &tables.TableMCPClient{}, "tool_execution_timeout") + }, + }}) + if err := m.Migrate(); err != nil { + return fmt.Errorf("error running %s migration: %w", migrationName, err) + } + return nil +} + +// migrationAddVirtualKeyExpiresAtColumn adds nullable expires_at to governance_virtual_keys. +// No index: expiry is checked in-memory from the already-loaded VK, never queried by column. +func migrationAddVirtualKeyExpiresAtColumn(ctx context.Context, db *gorm.DB, logger schemas.Logger) error { + migrationName := "add_virtual_key_expires_at_column" + logger.Info("[configstore] starting migration %s", migrationName) + defer logger.Info("[configstore] finished migration %s", migrationName) + m := migrator.New(db, migrator.DefaultOptions, []*migrator.Migration{{ + ID: migrationName, + Migrate: func(tx *gorm.DB) error { + tx = tx.WithContext(ctx) + return addColumnIfNotExists(tx, logger, &tables.TableVirtualKey{}, "expires_at") + }, + Rollback: func(tx *gorm.DB) error { + tx = tx.WithContext(ctx) + return dropColumnIfExists(tx, logger, &tables.TableVirtualKey{}, "expires_at") + }, + }}) + if err := m.Migrate(); err != nil { + return fmt.Errorf("error running %s migration: %w", migrationName, err) + } + return nil +} diff --git a/framework/configstore/migrations_test.go b/framework/configstore/migrations_test.go index d3f4ef0f88d..71da431a22e 100644 --- a/framework/configstore/migrations_test.go +++ b/framework/configstore/migrations_test.go @@ -1167,6 +1167,7 @@ func TestTriggerMigrations_FreshDB(t *testing.T) { for _, table := range criticalTables { assert.True(t, migrator.HasTable(table), "table should exist: %T", table) } + assert.True(t, migrator.HasColumn(&tables.TableModelPricing{}, "is_deprecated"), "model pricing is_deprecated column should exist") } func TestTriggerMigrations_Idempotent(t *testing.T) { @@ -1249,7 +1250,7 @@ func TestFullMigration_VirtualKeyCRUD(t *testing.T) { vk := &tables.TableVirtualKey{ ID: "vk-test-001", Name: "test-virtual-key", - Value: "vk-secret-value-12345", + Value: *schemas.NewSecretVar("vk-secret-value-12345"), IsActive: bifrost.Ptr(true), CreatedAt: now, UpdatedAt: now, @@ -1265,7 +1266,7 @@ func TestFullMigration_VirtualKeyCRUD(t *testing.T) { assert.Equal(t, "vk-test-001", vks[0].ID) assert.Equal(t, "test-virtual-key", vks[0].Name) - assert.Equal(t, "vk-secret-value-12345", vks[0].Value) // AfterFind decrypts + assert.Equal(t, "vk-secret-value-12345", vks[0].Value.GetValue()) // AfterFind decrypts assert.True(t, vks[0].IsActiveValue()) // Verify encryption at raw DB level @@ -1395,7 +1396,7 @@ func TestFullMigration_EncryptPlaintextRows(t *testing.T) { var vk tables.TableVirtualKey err = db.Where("id = ?", "vk-plain-1").First(&vk).Error require.NoError(t, err) - assert.Equal(t, "vk-plain-secret", vk.Value) + assert.Equal(t, "vk-plain-secret", vk.Value.Val) } func TestFullMigration_EndToEnd(t *testing.T) { @@ -1437,7 +1438,7 @@ func TestFullMigration_EndToEnd(t *testing.T) { {"vk-2", "vk-beta", "vk-beta-secret"}, } { err := store.CreateVirtualKey(ctx, &tables.TableVirtualKey{ - ID: vk.id, Name: vk.name, Value: vk.value, + ID: vk.id, Name: vk.name, Value: *schemas.NewSecretVar(vk.value), IsActive: bifrost.Ptr(true), CreatedAt: now, UpdatedAt: now, }) require.NoError(t, err, "CreateVirtualKey %s", vk.name) diff --git a/framework/configstore/rdb.go b/framework/configstore/rdb.go index f6f59d6a573..524cf1b2cb5 100644 --- a/framework/configstore/rdb.go +++ b/framework/configstore/rdb.go @@ -2,9 +2,14 @@ package configstore import ( "context" + "crypto/rand" + "crypto/rsa" + "crypto/x509" "encoding/json" + "encoding/pem" "errors" "fmt" + "math" "sort" "strings" "sync/atomic" @@ -121,6 +126,10 @@ func lockBudgetOwner(ctx context.Context, txDB *gorm.DB, budget tables.TableBudg return nil } +func toolExecutionTimeoutDurationToStoredSeconds(timeout time.Duration) int { + return int(math.Ceil(timeout.Seconds())) +} + func toolSyncIntervalDurationToStoredSeconds(interval time.Duration) (int, error) { if interval < 0 { return 0, fmt.Errorf("tool_sync_interval must be non-negative, got %q", interval.String()) @@ -134,52 +143,54 @@ func toolSyncIntervalDurationToStoredSeconds(interval time.Duration) (int, error // schemaKeyFromTableKey converts a database key to a schema key. func schemaKeyFromTableKey(dbKey tables.TableKey) schemas.Key { return schemas.Key{ - ID: dbKey.KeyID, - Name: dbKey.Name, - Value: dbKey.Value, - Models: dbKey.Models, - BlacklistedModels: dbKey.BlacklistedModels, - Weight: getWeight(dbKey.Weight), - Enabled: dbKey.Enabled, - UseForBatchAPI: dbKey.UseForBatchAPI, - AzureKeyConfig: dbKey.AzureKeyConfig, - VertexKeyConfig: dbKey.VertexKeyConfig, - BedrockKeyConfig: dbKey.BedrockKeyConfig, - Aliases: dbKey.Aliases, - VLLMKeyConfig: dbKey.VLLMKeyConfig, - ReplicateKeyConfig: dbKey.ReplicateKeyConfig, - OllamaKeyConfig: dbKey.OllamaKeyConfig, - SGLKeyConfig: dbKey.SGLKeyConfig, - ConfigHash: dbKey.ConfigHash, - Status: schemas.KeyStatusType(dbKey.Status), - Description: dbKey.Description, + ID: dbKey.KeyID, + Name: dbKey.Name, + Value: dbKey.Value, + Models: dbKey.Models, + BlacklistedModels: dbKey.BlacklistedModels, + Weight: getWeight(dbKey.Weight), + Enabled: dbKey.Enabled, + UseForBatchAPI: dbKey.UseForBatchAPI, + AzureKeyConfig: dbKey.AzureKeyConfig, + VertexKeyConfig: dbKey.VertexKeyConfig, + BedrockKeyConfig: dbKey.BedrockKeyConfig, + BedrockMantleKeyConfig: dbKey.BedrockMantleKeyConfig, + Aliases: dbKey.Aliases, + VLLMKeyConfig: dbKey.VLLMKeyConfig, + ReplicateKeyConfig: dbKey.ReplicateKeyConfig, + OllamaKeyConfig: dbKey.OllamaKeyConfig, + SGLKeyConfig: dbKey.SGLKeyConfig, + ConfigHash: dbKey.ConfigHash, + Status: schemas.KeyStatusType(dbKey.Status), + Description: dbKey.Description, } } // tableKeyFromSchemaKey converts a schema key to a database key. func tableKeyFromSchemaKey(provider tables.TableProvider, key schemas.Key) (tables.TableKey, error) { dbKey := tables.TableKey{ - Provider: provider.Name, - ProviderID: provider.ID, - KeyID: key.ID, - Name: key.Name, - Value: key.Value, - Models: key.Models, - BlacklistedModels: key.BlacklistedModels, - Weight: &key.Weight, - Enabled: key.Enabled, - UseForBatchAPI: key.UseForBatchAPI, - AzureKeyConfig: key.AzureKeyConfig, - VertexKeyConfig: key.VertexKeyConfig, - BedrockKeyConfig: key.BedrockKeyConfig, - Aliases: key.Aliases, - VLLMKeyConfig: key.VLLMKeyConfig, - ReplicateKeyConfig: key.ReplicateKeyConfig, - OllamaKeyConfig: key.OllamaKeyConfig, - SGLKeyConfig: key.SGLKeyConfig, - ConfigHash: key.ConfigHash, - Status: string(key.Status), - Description: key.Description, + Provider: provider.Name, + ProviderID: provider.ID, + KeyID: key.ID, + Name: key.Name, + Value: key.Value, + Models: key.Models, + BlacklistedModels: key.BlacklistedModels, + Weight: &key.Weight, + Enabled: key.Enabled, + UseForBatchAPI: key.UseForBatchAPI, + AzureKeyConfig: key.AzureKeyConfig, + VertexKeyConfig: key.VertexKeyConfig, + BedrockKeyConfig: key.BedrockKeyConfig, + BedrockMantleKeyConfig: key.BedrockMantleKeyConfig, + Aliases: key.Aliases, + VLLMKeyConfig: key.VLLMKeyConfig, + ReplicateKeyConfig: key.ReplicateKeyConfig, + OllamaKeyConfig: key.OllamaKeyConfig, + SGLKeyConfig: key.SGLKeyConfig, + ConfigHash: key.ConfigHash, + Status: string(key.Status), + Description: key.Description, } if key.AzureKeyConfig != nil { @@ -267,6 +278,8 @@ func (s *RDBConfigStore) UpdateClientConfig(ctx context.Context, config *ClientC AllowPerRequestContentStorageOverride: config.AllowPerRequestContentStorageOverride, AllowPerRequestRawOverride: config.AllowPerRequestRawOverride, AllowDirectKeys: config.AllowDirectKeys, + MCPServerAuthMode: config.MCPServerAuthMode, + OAuth2ServerConfig: config.OAuth2ServerConfig, ConfigHash: config.ConfigHash, } // Delete existing client config and create new one in a transaction. @@ -532,6 +545,8 @@ func (s *RDBConfigStore) GetClientConfig(ctx context.Context) (*ClientConfig, er AllowPerRequestContentStorageOverride: dbConfig.AllowPerRequestContentStorageOverride, AllowPerRequestRawOverride: dbConfig.AllowPerRequestRawOverride, AllowDirectKeys: dbConfig.AllowDirectKeys, + MCPServerAuthMode: dbConfig.MCPServerAuthMode, + OAuth2ServerConfig: dbConfig.OAuth2ServerConfig, ConfigHash: dbConfig.ConfigHash, }, nil } @@ -686,27 +701,28 @@ func (s *RDBConfigStore) UpdateProvidersConfig(ctx context.Context, providers ma } } dbKey := tables.TableKey{ - Provider: dbProvider.Name, - ProviderID: dbProvider.ID, - KeyID: key.ID, - Name: key.Name, - Value: key.Value, - Models: key.Models, - BlacklistedModels: key.BlacklistedModels, - Weight: &key.Weight, - Enabled: key.Enabled, - UseForBatchAPI: key.UseForBatchAPI, - AzureKeyConfig: key.AzureKeyConfig, - VertexKeyConfig: key.VertexKeyConfig, - BedrockKeyConfig: key.BedrockKeyConfig, - Aliases: key.Aliases, - VLLMKeyConfig: key.VLLMKeyConfig, - ReplicateKeyConfig: key.ReplicateKeyConfig, - OllamaKeyConfig: key.OllamaKeyConfig, - SGLKeyConfig: key.SGLKeyConfig, - ConfigHash: keyHash, - Status: string(key.Status), - Description: key.Description, + Provider: dbProvider.Name, + ProviderID: dbProvider.ID, + KeyID: key.ID, + Name: key.Name, + Value: key.Value, + Models: key.Models, + BlacklistedModels: key.BlacklistedModels, + Weight: &key.Weight, + Enabled: key.Enabled, + UseForBatchAPI: key.UseForBatchAPI, + AzureKeyConfig: key.AzureKeyConfig, + VertexKeyConfig: key.VertexKeyConfig, + BedrockKeyConfig: key.BedrockKeyConfig, + BedrockMantleKeyConfig: key.BedrockMantleKeyConfig, + Aliases: key.Aliases, + VLLMKeyConfig: key.VLLMKeyConfig, + ReplicateKeyConfig: key.ReplicateKeyConfig, + OllamaKeyConfig: key.OllamaKeyConfig, + SGLKeyConfig: key.SGLKeyConfig, + ConfigHash: keyHash, + Status: string(key.Status), + Description: key.Description, } // Handle Azure config @@ -914,27 +930,28 @@ func (s *RDBConfigStore) UpdateProvider(ctx context.Context, provider schemas.Mo return fmt.Errorf("failed to generate key hash: %w", err) } dbKey := tables.TableKey{ - Provider: dbProvider.Name, - ProviderID: dbProvider.ID, - KeyID: key.ID, - Name: key.Name, - Value: key.Value, - Models: key.Models, - BlacklistedModels: key.BlacklistedModels, - Weight: &key.Weight, - Enabled: key.Enabled, - UseForBatchAPI: key.UseForBatchAPI, - AzureKeyConfig: key.AzureKeyConfig, - VertexKeyConfig: key.VertexKeyConfig, - BedrockKeyConfig: key.BedrockKeyConfig, - Aliases: key.Aliases, - VLLMKeyConfig: key.VLLMKeyConfig, - ReplicateKeyConfig: key.ReplicateKeyConfig, - OllamaKeyConfig: key.OllamaKeyConfig, - SGLKeyConfig: key.SGLKeyConfig, - ConfigHash: keyHash, - Status: string(key.Status), - Description: key.Description, + Provider: dbProvider.Name, + ProviderID: dbProvider.ID, + KeyID: key.ID, + Name: key.Name, + Value: key.Value, + Models: key.Models, + BlacklistedModels: key.BlacklistedModels, + Weight: &key.Weight, + Enabled: key.Enabled, + UseForBatchAPI: key.UseForBatchAPI, + AzureKeyConfig: key.AzureKeyConfig, + VertexKeyConfig: key.VertexKeyConfig, + BedrockKeyConfig: key.BedrockKeyConfig, + BedrockMantleKeyConfig: key.BedrockMantleKeyConfig, + Aliases: key.Aliases, + VLLMKeyConfig: key.VLLMKeyConfig, + ReplicateKeyConfig: key.ReplicateKeyConfig, + OllamaKeyConfig: key.OllamaKeyConfig, + SGLKeyConfig: key.SGLKeyConfig, + ConfigHash: keyHash, + Status: string(key.Status), + Description: key.Description, } // Handle Azure config @@ -1053,27 +1070,28 @@ func (s *RDBConfigStore) AddProvider(ctx context.Context, provider schemas.Model // Create keys for this provider for _, key := range configCopy.Keys { dbKey := tables.TableKey{ - Provider: dbProvider.Name, - ProviderID: dbProvider.ID, - KeyID: key.ID, - Name: key.Name, - Value: key.Value, - Models: key.Models, - BlacklistedModels: key.BlacklistedModels, - Weight: &key.Weight, - Enabled: key.Enabled, - UseForBatchAPI: key.UseForBatchAPI, - AzureKeyConfig: key.AzureKeyConfig, - VertexKeyConfig: key.VertexKeyConfig, - BedrockKeyConfig: key.BedrockKeyConfig, - Aliases: key.Aliases, - VLLMKeyConfig: key.VLLMKeyConfig, - ReplicateKeyConfig: key.ReplicateKeyConfig, - OllamaKeyConfig: key.OllamaKeyConfig, - SGLKeyConfig: key.SGLKeyConfig, - ConfigHash: key.ConfigHash, - Status: string(key.Status), - Description: key.Description, + Provider: dbProvider.Name, + ProviderID: dbProvider.ID, + KeyID: key.ID, + Name: key.Name, + Value: key.Value, + Models: key.Models, + BlacklistedModels: key.BlacklistedModels, + Weight: &key.Weight, + Enabled: key.Enabled, + UseForBatchAPI: key.UseForBatchAPI, + AzureKeyConfig: key.AzureKeyConfig, + VertexKeyConfig: key.VertexKeyConfig, + BedrockKeyConfig: key.BedrockKeyConfig, + BedrockMantleKeyConfig: key.BedrockMantleKeyConfig, + Aliases: key.Aliases, + VLLMKeyConfig: key.VLLMKeyConfig, + ReplicateKeyConfig: key.ReplicateKeyConfig, + OllamaKeyConfig: key.OllamaKeyConfig, + SGLKeyConfig: key.SGLKeyConfig, + ConfigHash: key.ConfigHash, + Status: string(key.Status), + Description: key.Description, } // Handle Azure config if key.AzureKeyConfig != nil { @@ -1517,6 +1535,7 @@ func (s *RDBConfigStore) GetMCPConfig(ctx context.Context) (*schemas.MCPConfig, AllowedExtraHeaders: dbClient.AllowedExtraHeaders, IsPingAvailable: dbClient.IsPingAvailable, ToolSyncInterval: time.Duration(dbClient.ToolSyncInterval) * time.Second, + ToolExecutionTimeout: time.Duration(dbClient.ToolExecutionTimeout) * time.Second, ToolPricing: dbClient.ToolPricing, AllowOnAllVirtualKeys: dbClient.AllowOnAllVirtualKeys, Disabled: dbClient.Disabled, @@ -1559,6 +1578,7 @@ func (s *RDBConfigStore) GetMCPConfig(ctx context.Context) (*schemas.MCPConfig, AllowedExtraHeaders: dbClient.AllowedExtraHeaders, IsPingAvailable: dbClient.IsPingAvailable, ToolSyncInterval: time.Duration(dbClient.ToolSyncInterval) * time.Second, + ToolExecutionTimeout: time.Duration(dbClient.ToolExecutionTimeout) * time.Second, AllowOnAllVirtualKeys: dbClient.AllowOnAllVirtualKeys, Disabled: dbClient.Disabled, ToolPricing: dbClient.ToolPricing, @@ -1581,6 +1601,54 @@ func (s *RDBConfigStore) GetMCPClientsPaginated(ctx context.Context, params MCPC search := "%" + strings.ToLower(params.Search) + "%" baseQuery = baseQuery.Where("LOWER(name) LIKE ?", search) } + if params.ClientID != "" { + baseQuery = baseQuery.Where("client_id = ?", params.ClientID) + } + if len(params.ConnectionTypes) > 0 { + baseQuery = baseQuery.Where("connection_type IN ?", params.ConnectionTypes) + } + if len(params.AuthTypes) > 0 { + baseQuery = baseQuery.Where("auth_type IN ?", params.AuthTypes) + } + if params.IsCodeModeClient != nil { + baseQuery = baseQuery.Where("is_code_mode_client = ?", *params.IsCodeModeClient) + } + if params.Disabled != nil { + baseQuery = baseQuery.Where("disabled = ?", *params.Disabled) + } + // Runtime state filter, resolved by the caller into a connected-id set. + if params.StateInclude != nil { + if *params.StateInclude { + // connected: must be in the connected set. An empty set (nothing + // connected) yields IN (NULL) → matches no rows, which is correct. + baseQuery = baseQuery.Where("client_id IN ?", params.StateClientIDs) + } else if len(params.StateClientIDs) > 0 { + // disconnected: everything not currently connected. An empty + // connected set means all rows are disconnected → no constraint. + baseQuery = baseQuery.Where("client_id NOT IN ?", params.StateClientIDs) + } + } + // VK access filter: OR the "open to all VKs" flag with an explicit-assignment + // subquery over the VK⇄MCP join table (matched on the numeric primary key). + if params.OnlyAllVirtualKeys || len(params.VirtualKeyIDs) > 0 { + var assignedSub *gorm.DB + if len(params.VirtualKeyIDs) > 0 { + assignedSub = s.DB().WithContext(ctx). + Model(&tables.TableVirtualKeyMCPConfig{}). + Select("mcp_client_id"). + Where("virtual_key_id IN ?", params.VirtualKeyIDs) + } + switch { + case params.OnlyAllVirtualKeys && assignedSub != nil: + baseQuery = baseQuery.Where( + s.DB().Where("allow_on_all_virtual_keys = ?", true).Or("id IN (?)", assignedSub), + ) + case params.OnlyAllVirtualKeys: + baseQuery = baseQuery.Where("allow_on_all_virtual_keys = ?", true) + default: + baseQuery = baseQuery.Where("id IN (?)", assignedSub) + } + } var totalCount int64 if err := baseQuery.Count(&totalCount).Error; err != nil { @@ -1933,6 +2001,7 @@ func (s *RDBConfigStore) GetMCPClientConfigByID(ctx context.Context, id string) AllowedExtraHeaders: dbClient.AllowedExtraHeaders, IsPingAvailable: dbClient.IsPingAvailable, ToolSyncInterval: time.Duration(dbClient.ToolSyncInterval) * time.Second, + ToolExecutionTimeout: time.Duration(dbClient.ToolExecutionTimeout) * time.Second, AllowOnAllVirtualKeys: dbClient.AllowOnAllVirtualKeys, Disabled: dbClient.Disabled, ToolPricing: dbClient.ToolPricing, @@ -1971,6 +2040,7 @@ func (s *RDBConfigStore) CreateMCPClientConfig(ctx context.Context, clientConfig if err != nil { return err } + toolExecutionTimeoutSec := toolExecutionTimeoutDurationToStoredSeconds(clientConfigCopy.ToolExecutionTimeout) dbClient := tables.TableMCPClient{ ClientID: clientConfigCopy.ID, Name: clientConfigCopy.Name, @@ -1987,6 +2057,7 @@ func (s *RDBConfigStore) CreateMCPClientConfig(ctx context.Context, clientConfig AllowedExtraHeaders: clientConfigCopy.AllowedExtraHeaders, IsPingAvailable: clientConfigCopy.IsPingAvailable, ToolSyncInterval: toolSyncIntervalSec, + ToolExecutionTimeout: toolExecutionTimeoutSec, AllowOnAllVirtualKeys: clientConfigCopy.AllowOnAllVirtualKeys, // DiscoveredTools has json:"-" so deepCopy loses it; use original clientConfig DiscoveredTools: clientConfig.DiscoveredTools, @@ -2127,6 +2198,10 @@ func (s *RDBConfigStore) UpdateMCPClientConfig(ctx context.Context, id string, c // Update only editable fields using a map to avoid updating connection info // Connection info (ConnectionType, ConnectionString, StdioConfig) is read-only and should not be modified via API + if clientConfigCopy.ToolExecutionTimeout < 0 { + return fmt.Errorf("tool_execution_timeout must be non-negative, got %d", clientConfigCopy.ToolExecutionTimeout) + } + updates := map[string]interface{}{ "name": clientConfigCopy.Name, "is_code_mode_client": clientConfigCopy.IsCodeModeClient, @@ -2136,6 +2211,7 @@ func (s *RDBConfigStore) UpdateMCPClientConfig(ctx context.Context, id string, c "allowed_extra_headers_json": string(allowedExtraHeadersJSON), "tool_pricing_json": string(toolPricingJSON), "tool_sync_interval": clientConfigCopy.ToolSyncInterval, + "tool_execution_timeout": clientConfigCopy.ToolExecutionTimeout, "allow_on_all_virtual_keys": clientConfigCopy.AllowOnAllVirtualKeys, "disabled": clientConfigCopy.Disabled, "updated_at": time.Now(), @@ -2372,6 +2448,7 @@ var pricingSyncUpdateColumns = []string{ "max_input_tokens", "max_output_tokens", "architecture", + "is_deprecated", // Costs - Text "input_cost_per_token", "output_cost_per_token", @@ -2739,6 +2816,51 @@ func (s *RDBConfigStore) UpsertModelParameters(ctx context.Context, params *tabl return nil } +const modelParametersUpsertBatchSize = 100 + +// UpsertModelParametersBatch inserts or updates model parameters in batches. +// The sync path uses this to avoid one DB round-trip per model parameter row. +func (s *RDBConfigStore) UpsertModelParametersBatch(ctx context.Context, params []tables.TableModelParameters, tx ...*gorm.DB) error { + if len(params) == 0 { + return nil + } + deduped := make([]tables.TableModelParameters, 0, len(params)) + seen := make(map[string]int, len(params)) + for _, param := range params { + if idx, ok := seen[param.Model]; ok { + deduped[idx] = param + continue + } + seen[param.Model] = len(deduped) + deduped = append(deduped, param) + } + var txDB *gorm.DB + if len(tx) > 0 { + txDB = tx[0] + } else { + txDB = s.DB() + } + db := txDB.WithContext(ctx) + + onConflict := clause.OnConflict{ + Columns: []clause.Column{{Name: "model"}}, + UpdateAll: true, + } + upsert := func(tx *gorm.DB) error { + // Unlike TableModelPricing, TableModelParameters has no nullable default + // columns, so GORM's multi-row INSERT does not emit DEFAULT values that + // SQLite rejects. + if err := tx.Clauses(onConflict).CreateInBatches(deduped, modelParametersUpsertBatchSize).Error; err != nil { + return s.parseGormError(err) + } + return nil + } + if len(tx) > 0 { + return upsert(db) + } + return db.Transaction(upsert) +} + // PLUGINS METHODS func (s *RDBConfigStore) GetPlugins(ctx context.Context) ([]*tables.TablePlugin, error) { @@ -3177,6 +3299,9 @@ func (s *RDBConfigStore) GetVirtualKeyByValue(ctx context.Context, value string) // Use hash-based lookup if hash column is populated, fall back to plaintext for backward compat if err := query.Where("value_hash = ?", valueHash).First(&virtualKey).Error; err != nil { if errors.Is(err, gorm.ErrRecordNotFound) { + if schemas.IsSecretRef(value) { + return nil, ErrNotFound + } // Fallback: try plaintext lookup for rows not yet migrated if err := query.Where("value = ?", value).First(&virtualKey).Error; err != nil { if errors.Is(err, gorm.ErrRecordNotFound) { @@ -3204,6 +3329,9 @@ func (s *RDBConfigStore) GetVirtualKeyQuotaByValue(ctx context.Context, value st Preload("ProviderConfigs.RateLimit") if err := baseQuery.Session(&gorm.Session{}).Where("value_hash = ?", valueHash).First(&virtualKey).Error; err != nil { if errors.Is(err, gorm.ErrRecordNotFound) { + if schemas.IsSecretRef(value) { + return nil, ErrNotFound + } // Fallback: try plaintext lookup for rows not yet migrated if err := baseQuery.Session(&gorm.Session{}).Where("value = ?", value).First(&virtualKey).Error; err != nil { if errors.Is(err, gorm.ErrRecordNotFound) { @@ -3259,7 +3387,7 @@ func (s *RDBConfigStore) UpdateVirtualKey(ctx context.Context, virtualKey *table } else { virtualKey.ID = existing.ID if err := txDB.WithContext(ctx). - Select("name", "description", "value", "is_active", "team_id", "customer_id", "rate_limit_id", "calendar_aligned", "config_hash", "updated_at", "encryption_status", "value_hash"). + Select("name", "description", "value", "is_active", "expires_at", "team_id", "customer_id", "rate_limit_id", "calendar_aligned", "config_hash", "updated_at", "encryption_status", "value_hash"). Updates(virtualKey).Error; err != nil { return s.parseGormError(err) } @@ -3381,6 +3509,16 @@ func (s *RDBConfigStore) DeleteVirtualKey(ctx context.Context, id string, tx ... if err := txDB.WithContext(ctx).Where("virtual_key_id = ?", id).Delete(&tables.TableOauthUserToken{}).Error; err != nil { return err } + // Revoke gateway-issued OAuth2 grants bound to this VK (vk-mode tokens are + // keyed by vk_id in bf_sub). They are revoked rather than deleted so they + // stop minting access tokens on refresh and drop off the active-grants + // view, while remaining available for reuse detection until the sweep. + now := time.Now() + if err := txDB.WithContext(ctx).Model(&tables.TableOAuth2RefreshToken{}). + Where("bf_mode = ? AND bf_sub = ? AND revoked_at IS NULL", string(schemas.MCPAuthModeVK), id). + Update("revoked_at", &now).Error; err != nil { + return err + } // Delete per-user MCP header credentials tied to this VK if err := txDB.WithContext(ctx).Where("virtual_key_id = ?", id).Delete(&tables.TableMCPPerUserHeaderCredential{}).Error; err != nil { return err @@ -5734,8 +5872,30 @@ func (s *RDBConfigStore) DeleteOauthToken(ctx context.Context, id string) error // GetExpiringOauthTokens retrieves tokens that are expiring before the given time func (s *RDBConfigStore) GetExpiringOauthTokens(ctx context.Context, before time.Time) ([]*tables.TableOauthToken, error) { var tokens []*tables.TableOauthToken + // Exclude tokens whose owning oauth_config has already reached a terminal + // state — "expired" (set when a refresh is permanently rejected, e.g. + // invalid_grant / Grant not found) or "revoked". Without this, the refresh + // worker re-selects a permanently-dead token on every tick (its expires_at + // stays in the past) and logs the same failure indefinitely; a dead grant + // needs re-authorization, not perpetual retries. + // + // Refresh is also limited to tokens whose oauth_config is referenced by + // at least one enabled MCP client: nothing consumes a token while every + // client using it is disabled (or gone), so background refresh would keep + // calling the identity provider forever for an unused connection. When a + // client is re-enabled or attached later, GetAccessToken refreshes inline + // on first use. result := s.DB().WithContext(ctx). Where("expires_at IS NOT NULL AND expires_at < ?", before). + Where("NOT EXISTS (?)", + s.DB().Model(&tables.TableOauthConfig{}). + Select("1"). + Where("oauth_configs.token_id = oauth_tokens.id AND oauth_configs.status IN ?", []string{"expired", "revoked"})). + Where("EXISTS (?)", + s.DB().Model(&tables.TableMCPClient{}). + Select("1"). + Joins("JOIN oauth_configs ON oauth_configs.id = config_mcp_clients.oauth_config_id"). + Where("oauth_configs.token_id = oauth_tokens.id AND config_mcp_clients.disabled = ?", false)). Find(&tokens) if result.Error != nil { return nil, fmt.Errorf("failed to get expiring tokens: %w", result.Error) @@ -6446,6 +6606,16 @@ func applyMCPSessionFilters(query *gorm.DB, params MCPSessionsFilterParams, t mc if len(params.MCPClientIDs) > 0 { query = query.Where(t.table+".mcp_client_id IN ?", params.MCPClientIDs) } + if params.Identity != "" { + // Exact match against whichever identity column carries the value for this + // row's mode. Parenthesized explicitly so the OR group ANDs cleanly with the + // filters above — GORM does not wrap raw-string conditions in parentheses, so + // without the parens the trailing ORs would escape the AND chain. + query = query.Where( + "("+t.table+".user_id = ? OR "+t.table+".virtual_key_id = ? OR "+t.table+".session_id = ?)", + params.Identity, params.Identity, params.Identity, + ) + } if params.Search != "" { needle := "%" + strings.ToLower(params.Search) + "%" query = query. @@ -6728,3 +6898,466 @@ func (s *RDBConfigStore) ReconcileMCPHeadersAfterMCPChange(ctx context.Context, return nil }) } + +// GetOAuth2SigningKey returns the signing key, creating and persisting a new +// RS2048 keypair if none exists yet. +func (s *RDBConfigStore) GetOAuth2SigningKey(ctx context.Context) (*tables.OAuth2SigningKey, error) { + key, err := s.loadOAuth2SigningKey(ctx) + if err != nil { + if errors.Is(err, ErrNotFound) { + // No key persisted yet — generate and store one atomically. + return s.createOAuth2SigningKey(ctx) + } + return nil, err + } + return key, nil +} + +// loadOAuth2SigningKey reads and decrypts the persisted signing key. It returns +// ErrNotFound when no key has been generated yet. +func (s *RDBConfigStore) loadOAuth2SigningKey(ctx context.Context) (*tables.OAuth2SigningKey, error) { + row, err := s.GetConfig(ctx, tables.GovernanceConfigKeyOAuth2SigningKey) + if err != nil { + if errors.Is(err, ErrNotFound) { + return nil, ErrNotFound + } + return nil, fmt.Errorf("get oauth2 signing key: %w", err) + } + if row == nil || row.Value == "" { + return nil, ErrNotFound + } + var key tables.OAuth2SigningKey + if err := json.Unmarshal([]byte(row.Value), &key); err != nil { + return nil, fmt.Errorf("unmarshal oauth2 signing key: %w", err) + } + // Decrypt off the stored marker, not the live encrypt.IsEnabled() flag, so a + // key persisted while encryption was disabled is not mangled once encryption + // is later turned on (mirrors the AfterFind hooks on secret-bearing tables). + if err := key.Decrypt(); err != nil { + return nil, err + } + return &key, nil +} + +func (s *RDBConfigStore) createOAuth2SigningKey(ctx context.Context) (*tables.OAuth2SigningKey, error) { + priv, err := rsa.GenerateKey(rand.Reader, 2048) + if err != nil { + return nil, fmt.Errorf("generate RSA key: %w", err) + } + + privBytes, err := x509.MarshalPKCS8PrivateKey(priv) + if err != nil { + return nil, fmt.Errorf("marshal RSA private key: %w", err) + } + privPEM := string(pem.EncodeToMemory(&pem.Block{Type: "PRIVATE KEY", Bytes: privBytes})) + + pubBytes, err := x509.MarshalPKIXPublicKey(&priv.PublicKey) + if err != nil { + return nil, fmt.Errorf("marshal RSA public key: %w", err) + } + pubPEM := string(pem.EncodeToMemory(&pem.Block{Type: "PUBLIC KEY", Bytes: pubBytes})) + + key := &tables.OAuth2SigningKey{ + KID: uuid.New().String(), + PrivateKeyPEM: privPEM, + PublicKeyPEM: pubPEM, + } + + // Encrypt the private key in place and stamp EncryptionStatus before storage + // (mirrors the BeforeSave hooks on secret-bearing tables). + if err := key.Encrypt(); err != nil { + return nil, err + } + + data, err := json.Marshal(key) + if err != nil { + return nil, fmt.Errorf("marshal oauth2 signing key: %w", err) + } + + // Persist atomically with INSERT ... ON CONFLICT DO NOTHING so concurrent + // first-use callers cannot last-writer-wins different keypairs. If the row + // already exists (RowsAffected == 0), another caller won the race — reload + // and return the persisted key so every caller agrees on a single keypair. + res := s.DB().WithContext(ctx).Clauses(clause.OnConflict{ + Columns: []clause.Column{{Name: "key"}}, + DoNothing: true, + }).Create(&tables.TableGovernanceConfig{ + Key: tables.GovernanceConfigKeyOAuth2SigningKey, + Value: string(data), + }) + if res.Error != nil { + return nil, fmt.Errorf("persist oauth2 signing key: %w", res.Error) + } + if res.RowsAffected == 0 { + return s.loadOAuth2SigningKey(ctx) + } + + // We won the insert — return with plaintext private key for immediate use. + key.PrivateKeyPEM = privPEM + return key, nil +} + +// --- OAuth2 Clients (DCR) --- + +// CreateOAuth2Client persists a new DCR registration. +func (s *RDBConfigStore) CreateOAuth2Client(ctx context.Context, client *tables.TableOAuth2Client) error { + if err := s.DB().WithContext(ctx).Create(client).Error; err != nil { + return fmt.Errorf("create oauth2 client: %w", err) + } + return nil +} + +// GetOAuth2ClientByClientID returns the client with the given client_id, or nil +// if not found. +func (s *RDBConfigStore) GetOAuth2ClientByClientID(ctx context.Context, clientID string) (*tables.TableOAuth2Client, error) { + var c tables.TableOAuth2Client + err := s.DB().WithContext(ctx).Where("client_id = ?", clientID).First(&c).Error + if errors.Is(err, gorm.ErrRecordNotFound) { + return nil, ErrNotFound + } + if err != nil { + return nil, fmt.Errorf("get oauth2 client: %w", err) + } + return &c, nil +} + +// --- OAuth2 Authorize Requests --- + +// CreateOAuth2AuthorizeRequest persists a new pending authorize request. +func (s *RDBConfigStore) CreateOAuth2AuthorizeRequest(ctx context.Context, req *tables.TableOAuth2AuthorizeRequest) error { + if err := s.DB().WithContext(ctx).Create(req).Error; err != nil { + return fmt.Errorf("create oauth2 authorize request: %w", err) + } + return nil +} + +// GetOAuth2AuthorizeRequestByID returns the authorize request with the given ID. +func (s *RDBConfigStore) GetOAuth2AuthorizeRequestByID(ctx context.Context, id string) (*tables.TableOAuth2AuthorizeRequest, error) { + var req tables.TableOAuth2AuthorizeRequest + err := s.DB().WithContext(ctx).Where("id = ?", id).First(&req).Error + if errors.Is(err, gorm.ErrRecordNotFound) { + return nil, ErrNotFound + } + if err != nil { + return nil, fmt.Errorf("get oauth2 authorize request: %w", err) + } + return &req, nil +} + +// GetOAuth2AuthorizeRequestByCodeHash finds a consented authorize request by +// the hash of the auth code. Used by the token endpoint. +func (s *RDBConfigStore) GetOAuth2AuthorizeRequestByCodeHash(ctx context.Context, codeHash string) (*tables.TableOAuth2AuthorizeRequest, error) { + var req tables.TableOAuth2AuthorizeRequest + err := s.DB().WithContext(ctx). + Where("code_hash = ? AND status = ?", codeHash, tables.OAuth2AuthorizeRequestStatusConsented). + First(&req).Error + if errors.Is(err, gorm.ErrRecordNotFound) { + return nil, ErrNotFound + } + if err != nil { + return nil, fmt.Errorf("get oauth2 authorize request by code hash: %w", err) + } + return &req, nil +} + +// ConsentOAuth2AuthorizeRequest atomically transitions a still-pending authorize +// request to consented, recording the minted code hash and resolved identity in +// a single conditional update. The status guard makes the transition idempotent +// under concurrency: a second consent for the same flow matches zero rows and +// returns ErrNotFound rather than overwriting the code hash the first one minted. +func (s *RDBConfigStore) ConsentOAuth2AuthorizeRequest(ctx context.Context, req *tables.TableOAuth2AuthorizeRequest) error { + result := s.DB().WithContext(ctx).Model(&tables.TableOAuth2AuthorizeRequest{}). + Where("id = ? AND status = ?", req.ID, tables.OAuth2AuthorizeRequestStatusPending). + Updates(map[string]any{ + "status": tables.OAuth2AuthorizeRequestStatusConsented, + "code_hash": req.CodeHash, + "bf_mode": req.BfMode, + "bf_sub": req.BfSub, + "updated_at": req.UpdatedAt, + }) + if result.Error != nil { + return fmt.Errorf("consent authorize request: %w", result.Error) + } + if result.RowsAffected == 0 { + return ErrNotFound + } + return nil +} + +// SweepExpiredOAuth2AuthorizeRequests deletes pending/consented requests past +// their TTL. Safe to call periodically. +func (s *RDBConfigStore) SweepExpiredOAuth2AuthorizeRequests(ctx context.Context) error { + return s.DB().WithContext(ctx). + Where("expires_at < ? AND status != ?", time.Now(), tables.OAuth2AuthorizeRequestStatusCodeIssued). + Delete(&tables.TableOAuth2AuthorizeRequest{}).Error +} + +// --- OAuth2 Refresh Tokens --- + +// GetOAuth2RefreshTokenByHash returns the refresh token row for the given hash. +func (s *RDBConfigStore) GetOAuth2RefreshTokenByHash(ctx context.Context, hash string) (*tables.TableOAuth2RefreshToken, error) { + var rt tables.TableOAuth2RefreshToken + err := s.DB().WithContext(ctx).Where("token_hash = ? AND revoked_at IS NULL", hash).First(&rt).Error + if errors.Is(err, gorm.ErrRecordNotFound) { + return nil, ErrNotFound + } + if err != nil { + return nil, fmt.Errorf("get oauth2 refresh token: %w", err) + } + return &rt, nil +} + +// ConsumeOAuth2AuthorizeRequest atomically marks the authorize request as +// code_issued and creates the refresh token in a single transaction. +// If either operation fails the transaction is rolled back — the authorize +// request stays in "consented" state and the client can retry the token exchange. +func (s *RDBConfigStore) ConsumeOAuth2AuthorizeRequest(ctx context.Context, requestID string, rt *tables.TableOAuth2RefreshToken) error { + now := time.Now() + return s.DB().WithContext(ctx).Transaction(func(tx *gorm.DB) error { + // Conditional update guards single-use: only a still-consented, unexpired + // request transitions. A zero-row result means the code was already + // consumed, expired, or never consented — reject before minting a token so + // a racing second exchange can't double-spend one authorization code. + result := tx.Model(&tables.TableOAuth2AuthorizeRequest{}). + Where("id = ? AND status = ? AND expires_at > ?", requestID, tables.OAuth2AuthorizeRequestStatusConsented, now). + Updates(map[string]any{ + "status": tables.OAuth2AuthorizeRequestStatusCodeIssued, + "updated_at": now, + }) + if result.Error != nil { + return fmt.Errorf("consume authorize request: %w", result.Error) + } + if result.RowsAffected == 0 { + return ErrNotFound + } + if err := tx.Create(rt).Error; err != nil { + return fmt.Errorf("create refresh token: %w", err) + } + return nil + }) +} + +// RotateOAuth2RefreshToken atomically revokes the old refresh token and creates +// the new one in a single transaction. If either operation fails the transaction +// is rolled back — the old token stays active and the client can retry the refresh. +func (s *RDBConfigStore) RotateOAuth2RefreshToken(ctx context.Context, oldID string, newRT *tables.TableOAuth2RefreshToken) error { + now := time.Now() + return s.DB().WithContext(ctx).Transaction(func(tx *gorm.DB) error { + // Only an active (not-yet-revoked) token may be rotated. A zero-row result + // means the token was already revoked — either by a concurrent rotation or + // as a replay — so reject before minting a replacement. + result := tx.Model(&tables.TableOAuth2RefreshToken{}). + Where("id = ? AND revoked_at IS NULL", oldID). + Update("revoked_at", &now) + if result.Error != nil { + return fmt.Errorf("revoke old refresh token: %w", result.Error) + } + if result.RowsAffected == 0 { + return ErrNotFound + } + if err := tx.Create(newRT).Error; err != nil { + return fmt.Errorf("create new refresh token: %w", err) + } + return nil + }) +} + +// GetOAuth2RefreshTokenByHashAny returns a refresh token row including revoked +// ones. Used for stolen-token detection: if a revoked token is presented we can +// identify its family and revoke all descendants (RFC 9700 §2.2.2). +func (s *RDBConfigStore) GetOAuth2RefreshTokenByHashAny(ctx context.Context, hash string) (*tables.TableOAuth2RefreshToken, error) { + var rt tables.TableOAuth2RefreshToken + err := s.DB().WithContext(ctx).Where("token_hash = ?", hash).First(&rt).Error + if errors.Is(err, gorm.ErrRecordNotFound) { + return nil, ErrNotFound + } + if err != nil { + return nil, fmt.Errorf("get oauth2 refresh token (any): %w", err) + } + return &rt, nil +} + +// RevokeOAuth2RefreshTokensByFamilyID revokes all active refresh tokens sharing +// the same family ID. Called when a revoked token is re-presented, indicating +// the token family has been compromised (RFC 9700 §2.2.2). +func (s *RDBConfigStore) RevokeOAuth2RefreshTokensByFamilyID(ctx context.Context, familyID string) error { + now := time.Now() + return s.DB().WithContext(ctx). + Model(&tables.TableOAuth2RefreshToken{}). + Where("family_id = ? AND revoked_at IS NULL", familyID). + Update("revoked_at", &now).Error +} + +// RevokeOAuth2RefreshTokensByMode revokes all active refresh tokens for a given +// bf_mode — a bulk-revoke utility for invalidating every grant of one identity +// type (vk, user, or session). +func (s *RDBConfigStore) RevokeOAuth2RefreshTokensByMode(ctx context.Context, bfMode string) error { + now := time.Now() + return s.DB().WithContext(ctx). + Model(&tables.TableOAuth2RefreshToken{}). + Where("bf_mode = ? AND revoked_at IS NULL", bfMode). + Update("revoked_at", &now).Error +} + +// SweepOAuth2RefreshTokens deletes revoked refresh tokens older than the given +// duration. Active tokens are never swept — only revoked ones that are past +// their retention window. +func (s *RDBConfigStore) SweepOAuth2RefreshTokens(ctx context.Context, revokedOlderThan time.Duration) (int64, error) { + // A non-positive retention would put the cutoff at (or after) now, deleting + // nearly every revoked token — including the rows kept for stolen-token replay + // detection. Treat it as "don't sweep" rather than wiping the retention window. + if revokedOlderThan <= 0 { + return 0, nil + } + cutoff := time.Now().Add(-revokedOlderThan) + result := s.DB().WithContext(ctx). + Where("revoked_at IS NOT NULL AND revoked_at < ?", cutoff). + Delete(&tables.TableOAuth2RefreshToken{}) + return result.RowsAffected, result.Error +} + +// SweepOrphanedOAuth2Clients deletes dynamically-registered clients that no +// longer back any refresh token and were registered before the grace cutoff. +// +// Clients are minted per OAuth flow via Dynamic Client Registration, so they +// accumulate unbounded without this. A client is safe to drop once it owns no +// refresh token rows at all: either it never completed a flow (abandoned +// registration) or every grant it issued has since been revoked and aged out. +// This must run after the revoked-token sweep so a client whose tokens are +// still within their retention window — and thus still needed for refresh-token +// reuse detection — keeps its rows and is not collected prematurely. +// +// The grace cutoff protects a client mid-handshake (authorization code issued +// but not yet exchanged for tokens), which legitimately owns no tokens yet; +// registeredOlderThan must exceed the authorization code TTL. +func (s *RDBConfigStore) SweepOrphanedOAuth2Clients(ctx context.Context, registeredOlderThan time.Duration) (int64, error) { + cutoff := time.Now().Add(-registeredOlderThan) + result := s.DB().WithContext(ctx). + Where("created_at < ? AND NOT EXISTS (?)", cutoff, + s.DB().Model(&tables.TableOAuth2RefreshToken{}). + Select("1"). + Where("oauth2_refresh_tokens.client_id = oauth2_clients.client_id")). + Delete(&tables.TableOAuth2Client{}) + return result.RowsAffected, result.Error +} + +// OAuth2SessionRow is the wire shape for a single downstream grant in the +// Connected Clients list. +type OAuth2SessionRow struct { + ID string `json:"id"` + ClientID string `json:"client_id"` + ClientName string `json:"client_name,omitempty"` + BfMode string `json:"bf_mode"` + BfSub string `json:"bf_sub"` + BfSubDisplay string `json:"bf_sub_display,omitempty"` // human-readable: VK name for vk mode + Scope string `json:"scope"` + CreatedAt time.Time `json:"created_at"` + LastUsedAt *time.Time `json:"last_used_at,omitempty"` +} + +// ListOAuth2Sessions returns a page of active (non-revoked) refresh token rows, +// joined with their client names from oauth2_clients and VK names from +// governance_virtual_keys for vk-mode grants. Filtering (search + mode) and +// pagination (limit/offset) are pushed to SQL; the second return value is the +// total count matching the filters before the page slice. Uses ScopedDB so +// callers can inject row-visibility predicates via the context. +func (s *RDBConfigStore) ListOAuth2Sessions(ctx context.Context, params OAuth2SessionsQueryParams) ([]OAuth2SessionRow, int64, error) { + // base carries the joins + filters shared by the count and the page query. + // Session() forks it so Count and Find don't pollute each other's statement. + base := s.ScopedDB(ctx). + Table("oauth2_refresh_tokens rt"). + Joins("LEFT JOIN oauth2_clients c ON c.client_id = rt.client_id"). + Joins("LEFT JOIN governance_virtual_keys vk ON vk.id = rt.bf_sub AND rt.bf_mode = 'vk'"). + Where("rt.revoked_at IS NULL") + + if search := strings.TrimSpace(params.Search); search != "" { + like := "%" + strings.ToLower(search) + "%" + // Match the columns the UI renders: client name/id, the bound identity + // (bf_sub), and the joined VK name shown as the display name for vk mode. + base = base.Where( + "LOWER(COALESCE(c.client_name, '')) LIKE ? OR LOWER(rt.client_id) LIKE ? OR LOWER(rt.bf_sub) LIKE ? OR LOWER(COALESCE(vk.name, '')) LIKE ?", + like, like, like, like, + ) + } + if len(params.Modes) > 0 { + base = base.Where("rt.bf_mode IN ?", params.Modes) + } + + var totalCount int64 + if err := base.Session(&gorm.Session{}).Count(&totalCount).Error; err != nil { + return nil, 0, fmt.Errorf("count oauth2 sessions: %w", err) + } + + rows := []struct { + tables.TableOAuth2RefreshToken + ClientName string `gorm:"column:client_name"` + VKName string `gorm:"column:vk_name"` + }{} + query := base.Session(&gorm.Session{}). + Select("rt.*, c.client_name, vk.name as vk_name"). + // id is the unique tiebreaker: created_at alone is not unique, and offset + // paging over a non-unique sort lets same-timestamp rows shuffle between + // pages (duplicated or skipped). id pins a deterministic order. + Order("rt.created_at DESC"). + Order("rt.id DESC") + if params.Limit > 0 { + query = query.Limit(params.Limit) + } + if params.Offset > 0 { + query = query.Offset(params.Offset) + } + if err := query.Scan(&rows).Error; err != nil { + return nil, 0, fmt.Errorf("list oauth2 sessions: %w", err) + } + out := make([]OAuth2SessionRow, 0, len(rows)) + for _, r := range rows { + row := OAuth2SessionRow{ + ID: r.ID, + ClientID: r.ClientID, + ClientName: r.ClientName, + BfMode: r.BfMode, + BfSub: r.BfSub, + Scope: r.Scope, + CreatedAt: r.CreatedAt, + LastUsedAt: r.LastUsedAt, + } + // Populate human-readable display name for VK mode. + if r.BfMode == "vk" && r.VKName != "" { + row.BfSubDisplay = r.VKName + } + out = append(out, row) + } + return out, totalCount, nil +} + +// RevokeOAuth2Session revokes a specific refresh token by ID (for use from the +// Connected Clients UI). Returns ErrNotFound when the ID does not exist. +func (s *RDBConfigStore) RevokeOAuth2Session(ctx context.Context, id string) error { + now := time.Now() + result := s.DB().WithContext(ctx). + Model(&tables.TableOAuth2RefreshToken{}). + Where("id = ? AND revoked_at IS NULL", id). + Update("revoked_at", &now) + if result.Error != nil { + return fmt.Errorf("revoke oauth2 session: %w", result.Error) + } + if result.RowsAffected == 0 { + return ErrNotFound + } + return nil +} + +// GetOAuth2SessionByID returns a single active refresh token row by ID. +// Uses ScopedDB so row-visibility predicates injected into the context apply. +// Returns ErrNotFound when the ID does not exist or is already revoked. +func (s *RDBConfigStore) GetOAuth2SessionByID(ctx context.Context, id string) (*tables.TableOAuth2RefreshToken, error) { + var rt tables.TableOAuth2RefreshToken + err := s.ScopedDB(ctx).Where("id = ? AND revoked_at IS NULL", id).First(&rt).Error + if errors.Is(err, gorm.ErrRecordNotFound) { + return nil, ErrNotFound + } + if err != nil { + return nil, fmt.Errorf("get oauth2 session: %w", err) + } + return &rt, nil +} diff --git a/framework/configstore/rdb_deadlock_postgres_test.go b/framework/configstore/rdb_deadlock_postgres_test.go index a5cbc39d76c..794964ab81a 100644 --- a/framework/configstore/rdb_deadlock_postgres_test.go +++ b/framework/configstore/rdb_deadlock_postgres_test.go @@ -162,7 +162,7 @@ func TestPostgresVirtualKeyBudgetConcurrentMutationsDoNotDeadlock(t *testing.T) if err := store.UpdateVirtualKey(ctx, &tables.TableVirtualKey{ ID: vkID, Name: "PG VK Budget", - Value: "pg-vk-budget-value", + Value: *schemas.NewSecretVar("pg-vk-budget-value"), IsActive: schemas.Ptr(true), }, tx); err != nil { return err @@ -247,7 +247,7 @@ func seedProviderGraph(ctx context.Context, store *RDBConfigStore) error { if err := store.CreateVirtualKey(ctx, &tables.TableVirtualKey{ ID: "pg-vk", Name: "PG VK", - Value: fmt.Sprintf("pg-vk-value-%d", time.Now().UnixNano()), + Value: *schemas.NewSecretVar(fmt.Sprintf("pg-vk-value-%d", time.Now().UnixNano())), IsActive: schemas.Ptr(true), }); err != nil && !isUniqueRace(err) { return err @@ -271,7 +271,7 @@ func seedVirtualKeyBudget(ctx context.Context, t *testing.T, store *RDBConfigSto require.NoError(t, store.CreateVirtualKey(ctx, &tables.TableVirtualKey{ ID: vkID, Name: "PG VK Budget", - Value: "pg-vk-budget-value", + Value: *schemas.NewSecretVar("pg-vk-budget-value"), IsActive: schemas.Ptr(true), })) require.NoError(t, store.CreateBudget(ctx, &tables.TableBudget{ diff --git a/framework/configstore/rdb_mcp_sessions_identity_test.go b/framework/configstore/rdb_mcp_sessions_identity_test.go new file mode 100644 index 00000000000..95f23324e26 --- /dev/null +++ b/framework/configstore/rdb_mcp_sessions_identity_test.go @@ -0,0 +1,60 @@ +package configstore + +import ( + "context" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// The fixture seeds three tokens: tok-active (vk → virtual_key_id=vk-alpha), +// tok-orphan (user → user_id=user-42), tok-reauth (session → session_id=sess-xyz). + +func TestListOauthUserTokens_IdentityExactMatch(t *testing.T) { + store := setupMCPSessionsTestStore(t) + seedMCPSessionsFixture(t, store) + ctx := context.Background() + + cases := []struct { + identity string + wantID string + }{ + {"user-42", "tok-orphan"}, + {"vk-alpha", "tok-active"}, + {"sess-xyz", "tok-reauth"}, + } + for _, tc := range cases { + t.Run(tc.identity, func(t *testing.T) { + got, err := store.ListOauthUserTokens(ctx, MCPSessionsFilterParams{Identity: tc.identity}) + require.NoError(t, err) + require.Len(t, got, 1) + assert.Equal(t, tc.wantID, got[0].ID) + }) + } +} + +// TestListOauthUserTokens_IdentityComposesWithAuthMode pins the parenthesization +// of the identity OR group. With Identity=vk-alpha and AuthModes=[user], the +// only row whose virtual_key_id is vk-alpha is vk-mode, so the auth-mode filter +// must exclude it → zero rows. If the OR group were not parenthesized, the +// trailing `OR virtual_key_id = ?` would escape the AND chain and leak the +// vk-mode row back in. +func TestListOauthUserTokens_IdentityComposesWithAuthMode(t *testing.T) { + store := setupMCPSessionsTestStore(t) + seedMCPSessionsFixture(t, store) + ctx := context.Background() + + leaked, err := store.ListOauthUserTokens(ctx, MCPSessionsFilterParams{ + Identity: "vk-alpha", AuthModes: []string{"user"}, + }) + require.NoError(t, err) + assert.Empty(t, leaked, "auth_mode filter must AND with the whole identity OR group") + + matched, err := store.ListOauthUserTokens(ctx, MCPSessionsFilterParams{ + Identity: "vk-alpha", AuthModes: []string{"vk"}, + }) + require.NoError(t, err) + require.Len(t, matched, 1) + assert.Equal(t, "tok-active", matched[0].ID) +} diff --git a/framework/configstore/rdb_mcp_sessions_test.go b/framework/configstore/rdb_mcp_sessions_test.go index 5f25f7398fe..72fed8b1166 100644 --- a/framework/configstore/rdb_mcp_sessions_test.go +++ b/framework/configstore/rdb_mcp_sessions_test.go @@ -5,6 +5,7 @@ import ( "testing" "time" + "github.com/maximhq/bifrost/core/schemas" "github.com/maximhq/bifrost/framework/configstore/tables" "github.com/stretchr/testify/require" ) @@ -39,7 +40,7 @@ func seedMCPSessionsFixture(t *testing.T, store *RDBConfigStore) { vk := &tables.TableVirtualKey{ ID: "vk-alpha", Name: "Alpha VK", - Value: "sk-bf-alpha", + Value: *schemas.NewSecretVar("sk-bf-alpha"), } require.NoError(t, store.DB().WithContext(ctx).Create(vk).Error) diff --git a/framework/configstore/rdb_oauth2_test.go b/framework/configstore/rdb_oauth2_test.go new file mode 100644 index 00000000000..cbebcfb2583 --- /dev/null +++ b/framework/configstore/rdb_oauth2_test.go @@ -0,0 +1,613 @@ +package configstore + +import ( + "context" + "testing" + "time" + + "github.com/maximhq/bifrost/core/schemas" + "github.com/maximhq/bifrost/framework/configstore/tables" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// setupOAuth2TestStore extends the base in-memory store with the OAuth2 issuance +// tables, which are not part of the base migration set. +func setupOAuth2TestStore(t *testing.T) *RDBConfigStore { + t.Helper() + s := setupRDBTestStore(t) + require.NoError(t, s.DB().AutoMigrate( + &tables.TableOAuth2Client{}, + &tables.TableOAuth2AuthorizeRequest{}, + &tables.TableOAuth2RefreshToken{}, + )) + return s +} + +// seedAuthorizeRequest inserts a request in the given status with a future expiry. +func seedAuthorizeRequest(t *testing.T, s *RDBConfigStore, id string, status tables.OAuth2AuthorizeRequestStatus, codeHash *string, expires time.Time) { + t.Helper() + req := &tables.TableOAuth2AuthorizeRequest{ + ID: id, + ClientID: "client-1", + RedirectURI: "http://127.0.0.1/cb", + State: "state", + Scope: "mcp", + Resource: "https://bifrost.test/mcp", + CodeChallenge: "challenge", + CodeChallengeMethod: "S256", + Status: status, + CodeHash: codeHash, + ExpiresAt: expires, + CreatedAt: time.Now(), + UpdatedAt: time.Now(), + } + require.NoError(t, s.CreateOAuth2AuthorizeRequest(context.Background(), req)) +} + +// makeRefreshToken builds a refresh-token row with sensible defaults. +func makeRefreshToken(id, familyID, clientID, hash string) *tables.TableOAuth2RefreshToken { + return &tables.TableOAuth2RefreshToken{ + ID: id, + TokenHash: hash, + FamilyID: familyID, + ClientID: clientID, + BfMode: "vk", + BfSub: "vk-1", + Scope: "mcp", + Resource: "https://bifrost.test/mcp", + CreatedAt: time.Now(), + } +} + +// seedExpiringTokenFixtures installs the token/config/client helpers shared by +// the GetExpiringOauthTokens tests. Every token is created already-expired so +// only the config/client conditions decide whether it is selected. +func seedExpiringTokenFixtures(t *testing.T, s *RDBConfigStore) (mkToken func(id string), mkConfig func(id, tokenID, status, state string), mkClient func(name, oauthConfigID string, disabled bool)) { + t.Helper() + require.NoError(t, s.DB().AutoMigrate(&tables.TableOauthConfig{}, &tables.TableOauthToken{}, &tables.TableMCPClient{})) + past := time.Now().Add(-time.Hour) + + mkToken = func(id string) { + require.NoError(t, s.DB().Create(&tables.TableOauthToken{ + ID: id, AccessToken: "at-" + id, TokenType: "Bearer", + ExpiresAt: &past, CreatedAt: time.Now(), UpdatedAt: time.Now(), + }).Error) + } + mkConfig = func(id, tokenID, status, state string) { + require.NoError(t, s.DB().Create(&tables.TableOauthConfig{ + ID: id, RedirectURI: "http://127.0.0.1/cb", State: state, Status: status, + TokenID: &tokenID, CreatedAt: time.Now(), UpdatedAt: time.Now(), + ExpiresAt: time.Now().Add(time.Hour), + }).Error) + } + mkClient = func(name, oauthConfigID string, disabled bool) { + require.NoError(t, s.DB().Create(&tables.TableMCPClient{ + ClientID: "cid-" + name, Name: name, ConnectionType: "http", + AuthType: "oauth", OauthConfigID: &oauthConfigID, Disabled: disabled, + CreatedAt: time.Now(), UpdatedAt: time.Now(), + }).Error) + } + return mkToken, mkConfig, mkClient +} + +func expiringTokenIDs(t *testing.T, s *RDBConfigStore) map[string]bool { + t.Helper() + got, err := s.GetExpiringOauthTokens(context.Background(), time.Now().Add(time.Minute)) + require.NoError(t, err) + ids := make(map[string]bool, len(got)) + for _, tk := range got { + ids[tk.ID] = true + } + return ids +} + +// TestGetExpiringOauthTokens_ExcludesTerminalConfigs verifies the refresh worker +// query skips tokens whose oauth_config is already terminal (expired/revoked), so +// a permanently-dead grant is not retried — and re-logged — on every tick. Each +// config gets an enabled MCP client so status is the only deciding condition. +func TestGetExpiringOauthTokens_ExcludesTerminalConfigs(t *testing.T) { + s := setupRDBTestStore(t) + mkToken, mkConfig, mkClient := seedExpiringTokenFixtures(t, s) + + mkToken("tok-live") + mkConfig("cfg-live", "tok-live", "authorized", "state-live") + mkClient("client-live", "cfg-live", false) + mkToken("tok-expired") + mkConfig("cfg-expired", "tok-expired", "expired", "state-expired") + mkClient("client-expired", "cfg-expired", false) + mkToken("tok-revoked") + mkConfig("cfg-revoked", "tok-revoked", "revoked", "state-revoked") + mkClient("client-revoked", "cfg-revoked", false) + + ids := expiringTokenIDs(t, s) + assert.True(t, ids["tok-live"], "token with an authorized config and enabled client should be refreshed") + assert.False(t, ids["tok-expired"], "token with an expired config must be excluded") + assert.False(t, ids["tok-revoked"], "token with a revoked config must be excluded") +} + +// TestGetExpiringOauthTokens_RequiresEnabledClient verifies the refresh worker +// only keeps tokens warm while at least one enabled MCP client references the +// owning oauth_config. Disabled-only and unreferenced configs are skipped — +// their tokens catch up via GetAccessToken's inline refresh on next use. +func TestGetExpiringOauthTokens_RequiresEnabledClient(t *testing.T) { + s := setupRDBTestStore(t) + mkToken, mkConfig, mkClient := seedExpiringTokenFixtures(t, s) + + mkToken("tok-enabled") + mkConfig("cfg-enabled", "tok-enabled", "authorized", "state-enabled") + mkClient("client-enabled", "cfg-enabled", false) + + mkToken("tok-disabled") + mkConfig("cfg-disabled", "tok-disabled", "authorized", "state-disabled") + mkClient("client-disabled", "cfg-disabled", true) + + mkToken("tok-shared") + mkConfig("cfg-shared", "tok-shared", "authorized", "state-shared") + mkClient("client-shared-off", "cfg-shared", true) + mkClient("client-shared-on", "cfg-shared", false) + + mkToken("tok-no-client") + mkConfig("cfg-no-client", "tok-no-client", "authorized", "state-no-client") + + mkToken("tok-orphan") // no owning config at all + + ids := expiringTokenIDs(t, s) + assert.True(t, ids["tok-enabled"], "token with an enabled client should be refreshed") + assert.False(t, ids["tok-disabled"], "token referenced only by a disabled client must be excluded") + assert.True(t, ids["tok-shared"], "config shared with at least one enabled client should be refreshed") + assert.False(t, ids["tok-no-client"], "token whose config has no client rows must be excluded") + assert.False(t, ids["tok-orphan"], "token with no owning config must be excluded") +} + +func TestGetOAuth2SigningKey_AutoGeneratesAndIsStable(t *testing.T) { + s := setupOAuth2TestStore(t) + ctx := context.Background() + + first, err := s.GetOAuth2SigningKey(ctx) + require.NoError(t, err) + require.NotNil(t, first) + assert.NotEmpty(t, first.KID) + assert.NotEmpty(t, first.PrivateKeyPEM) + assert.NotEmpty(t, first.PublicKeyPEM) + + // A second call must return the same persisted key, not mint a new one. + second, err := s.GetOAuth2SigningKey(ctx) + require.NoError(t, err) + assert.Equal(t, first.KID, second.KID) +} + +func TestConsentOAuth2AuthorizeRequest_AtomicPendingTransition(t *testing.T) { + s := setupOAuth2TestStore(t) + ctx := context.Background() + seedAuthorizeRequest(t, s, "req-1", tables.OAuth2AuthorizeRequestStatusPending, nil, time.Now().Add(time.Minute)) + + req := &tables.TableOAuth2AuthorizeRequest{ + ID: "req-1", + CodeHash: strPtr("code-hash-1"), + BfMode: "vk", + BfSub: "vk-1", + UpdatedAt: time.Now(), + } + require.NoError(t, s.ConsentOAuth2AuthorizeRequest(ctx, req)) + + got, err := s.GetOAuth2AuthorizeRequestByID(ctx, "req-1") + require.NoError(t, err) + assert.Equal(t, tables.OAuth2AuthorizeRequestStatusConsented, got.Status) + require.NotNil(t, got.CodeHash) + assert.Equal(t, "code-hash-1", *got.CodeHash) + assert.Equal(t, "vk", got.BfMode) + assert.Equal(t, "vk-1", got.BfSub) + + // A second consent on the now-consented row matches zero rows: ErrNotFound, + // and the originally minted code hash is left untouched. + err = s.ConsentOAuth2AuthorizeRequest(ctx, &tables.TableOAuth2AuthorizeRequest{ + ID: "req-1", CodeHash: strPtr("code-hash-2"), UpdatedAt: time.Now(), + }) + assert.ErrorIs(t, err, ErrNotFound) + + got, err = s.GetOAuth2AuthorizeRequestByID(ctx, "req-1") + require.NoError(t, err) + assert.Equal(t, "code-hash-1", *got.CodeHash) +} + +func TestConsumeOAuth2AuthorizeRequest_SingleUse(t *testing.T) { + s := setupOAuth2TestStore(t) + ctx := context.Background() + seedAuthorizeRequest(t, s, "req-1", tables.OAuth2AuthorizeRequestStatusConsented, strPtr("ch"), time.Now().Add(time.Minute)) + + rt := makeRefreshToken("rt-1", "req-1", "client-1", "hash-1") + require.NoError(t, s.ConsumeOAuth2AuthorizeRequest(ctx, "req-1", rt)) + + got, err := s.GetOAuth2AuthorizeRequestByID(ctx, "req-1") + require.NoError(t, err) + assert.Equal(t, tables.OAuth2AuthorizeRequestStatusCodeIssued, got.Status) + stored, err := s.GetOAuth2RefreshTokenByHash(ctx, "hash-1") + require.NoError(t, err) + assert.Equal(t, "rt-1", stored.ID) + + // Reuse of the same code: the row is already code_issued, so the second + // exchange matches zero rows and no second token is minted. + rt2 := makeRefreshToken("rt-2", "req-1", "client-1", "hash-2") + err = s.ConsumeOAuth2AuthorizeRequest(ctx, "req-1", rt2) + assert.ErrorIs(t, err, ErrNotFound) + _, err = s.GetOAuth2RefreshTokenByHash(ctx, "hash-2") + assert.ErrorIs(t, err, ErrNotFound) +} + +func TestConsumeOAuth2AuthorizeRequest_ExpiredCodeRejected(t *testing.T) { + s := setupOAuth2TestStore(t) + ctx := context.Background() + seedAuthorizeRequest(t, s, "req-1", tables.OAuth2AuthorizeRequestStatusConsented, strPtr("ch"), time.Now().Add(-time.Minute)) + + rt := makeRefreshToken("rt-1", "req-1", "client-1", "hash-1") + err := s.ConsumeOAuth2AuthorizeRequest(ctx, "req-1", rt) + assert.ErrorIs(t, err, ErrNotFound) + _, err = s.GetOAuth2RefreshTokenByHash(ctx, "hash-1") + assert.ErrorIs(t, err, ErrNotFound) +} + +func TestRotateOAuth2RefreshToken_RotationAndReplayGuard(t *testing.T) { + s := setupOAuth2TestStore(t) + ctx := context.Background() + old := makeRefreshToken("rt-old", "fam-1", "client-1", "hash-old") + require.NoError(t, s.DB().WithContext(ctx).Create(old).Error) + + newRT := makeRefreshToken("rt-new", "fam-1", "client-1", "hash-new") + require.NoError(t, s.RotateOAuth2RefreshToken(ctx, "rt-old", newRT)) + + // Old token is now revoked (no longer returned by the active-only lookup) but + // the new one is active and carries the same family. + _, err := s.GetOAuth2RefreshTokenByHash(ctx, "hash-old") + assert.ErrorIs(t, err, ErrNotFound) + active, err := s.GetOAuth2RefreshTokenByHash(ctx, "hash-new") + require.NoError(t, err) + assert.Equal(t, "fam-1", active.FamilyID) + + revoked, err := s.GetOAuth2RefreshTokenByHashAny(ctx, "hash-old") + require.NoError(t, err) + require.NotNil(t, revoked.RevokedAt) + + // Replaying the already-revoked token cannot rotate again. + err = s.RotateOAuth2RefreshToken(ctx, "rt-old", makeRefreshToken("rt-x", "fam-1", "client-1", "hash-x")) + assert.ErrorIs(t, err, ErrNotFound) +} + +func TestRevokeOAuth2RefreshTokensByFamilyID(t *testing.T) { + s := setupOAuth2TestStore(t) + ctx := context.Background() + require.NoError(t, s.DB().Create(makeRefreshToken("a", "fam-1", "c", "ha")).Error) + require.NoError(t, s.DB().Create(makeRefreshToken("b", "fam-1", "c", "hb")).Error) + require.NoError(t, s.DB().Create(makeRefreshToken("c", "fam-2", "c", "hc")).Error) + + require.NoError(t, s.RevokeOAuth2RefreshTokensByFamilyID(ctx, "fam-1")) + + // fam-1 fully revoked; fam-2 untouched. + _, err := s.GetOAuth2RefreshTokenByHash(ctx, "ha") + assert.ErrorIs(t, err, ErrNotFound) + _, err = s.GetOAuth2RefreshTokenByHash(ctx, "hb") + assert.ErrorIs(t, err, ErrNotFound) + survivor, err := s.GetOAuth2RefreshTokenByHash(ctx, "hc") + require.NoError(t, err) + assert.Equal(t, "fam-2", survivor.FamilyID) +} + +func TestSweepOAuth2RefreshTokens(t *testing.T) { + s := setupOAuth2TestStore(t) + ctx := context.Background() + retention := time.Hour + + oldRevoked := time.Now().Add(-2 * time.Hour) + recentRevoked := time.Now().Add(-time.Minute) + + staleTok := makeRefreshToken("stale", "f", "c", "h-stale") + staleTok.RevokedAt = &oldRevoked + recentTok := makeRefreshToken("recent", "f", "c", "h-recent") + recentTok.RevokedAt = &recentRevoked + activeTok := makeRefreshToken("active", "f", "c", "h-active") + require.NoError(t, s.DB().Create(staleTok).Error) + require.NoError(t, s.DB().Create(recentTok).Error) + require.NoError(t, s.DB().Create(activeTok).Error) + + deleted, err := s.SweepOAuth2RefreshTokens(ctx, retention) + require.NoError(t, err) + assert.Equal(t, int64(1), deleted) + + // Stale revoked gone; recently-revoked and active survive (still needed for + // replay detection / use). + _, err = s.GetOAuth2RefreshTokenByHashAny(ctx, "h-stale") + assert.ErrorIs(t, err, ErrNotFound) + _, err = s.GetOAuth2RefreshTokenByHashAny(ctx, "h-recent") + require.NoError(t, err) + _, err = s.GetOAuth2RefreshTokenByHash(ctx, "h-active") + require.NoError(t, err) +} + +func TestSweepOAuth2RefreshTokens_NonPositiveRetentionIsNoop(t *testing.T) { + s := setupOAuth2TestStore(t) + ctx := context.Background() + revoked := time.Now().Add(-time.Hour) + tok := makeRefreshToken("r", "f", "c", "h") + tok.RevokedAt = &revoked + require.NoError(t, s.DB().Create(tok).Error) + + deleted, err := s.SweepOAuth2RefreshTokens(ctx, 0) + require.NoError(t, err) + assert.Equal(t, int64(0), deleted) + _, err = s.GetOAuth2RefreshTokenByHashAny(ctx, "h") + require.NoError(t, err, "non-positive retention must not delete the replay-detection window") +} + +func TestSweepOrphanedOAuth2Clients(t *testing.T) { + s := setupOAuth2TestStore(t) + ctx := context.Background() + grace := time.Hour + old := time.Now().Add(-2 * time.Hour) + recent := time.Now() + + // withToken: backs a refresh token row → kept regardless of age. + require.NoError(t, s.DB().Create(&tables.TableOAuth2Client{ + ID: "c-token", ClientID: "with-token", RedirectURIs: []string{"http://127.0.0.1/cb"}, + GrantTypes: []string{"authorization_code"}, CreatedAt: old, + }).Error) + require.NoError(t, s.DB().Create(makeRefreshToken("rt", "fam", "with-token", "h")).Error) + + // orphanOld: no tokens, registered before the grace cutoff → swept. + require.NoError(t, s.DB().Create(&tables.TableOAuth2Client{ + ID: "c-old", ClientID: "orphan-old", RedirectURIs: []string{"http://127.0.0.1/cb"}, + GrantTypes: []string{"authorization_code"}, CreatedAt: old, + }).Error) + + // orphanFresh: no tokens but mid-handshake (within grace) → kept. + require.NoError(t, s.DB().Create(&tables.TableOAuth2Client{ + ID: "c-fresh", ClientID: "orphan-fresh", RedirectURIs: []string{"http://127.0.0.1/cb"}, + GrantTypes: []string{"authorization_code"}, CreatedAt: recent, + }).Error) + + deleted, err := s.SweepOrphanedOAuth2Clients(ctx, grace) + require.NoError(t, err) + assert.Equal(t, int64(1), deleted) + + _, err = s.GetOAuth2ClientByClientID(ctx, "orphan-old") + assert.ErrorIs(t, err, ErrNotFound) + _, err = s.GetOAuth2ClientByClientID(ctx, "with-token") + require.NoError(t, err) + _, err = s.GetOAuth2ClientByClientID(ctx, "orphan-fresh") + require.NoError(t, err) +} + +func TestSweepExpiredOAuth2AuthorizeRequests(t *testing.T) { + s := setupOAuth2TestStore(t) + ctx := context.Background() + past := time.Now().Add(-time.Minute) + future := time.Now().Add(time.Minute) + + seedAuthorizeRequest(t, s, "pending-expired", tables.OAuth2AuthorizeRequestStatusPending, nil, past) + seedAuthorizeRequest(t, s, "consented-expired", tables.OAuth2AuthorizeRequestStatusConsented, strPtr("ch"), past) + seedAuthorizeRequest(t, s, "issued-expired", tables.OAuth2AuthorizeRequestStatusCodeIssued, strPtr("ch2"), past) + seedAuthorizeRequest(t, s, "pending-fresh", tables.OAuth2AuthorizeRequestStatusPending, nil, future) + + require.NoError(t, s.SweepExpiredOAuth2AuthorizeRequests(ctx)) + + // Expired pending/consented are gone; an expired code_issued row is retained + // (it represents a completed exchange), and a fresh pending row survives. + _, err := s.GetOAuth2AuthorizeRequestByID(ctx, "pending-expired") + assert.ErrorIs(t, err, ErrNotFound) + _, err = s.GetOAuth2AuthorizeRequestByID(ctx, "consented-expired") + assert.ErrorIs(t, err, ErrNotFound) + _, err = s.GetOAuth2AuthorizeRequestByID(ctx, "issued-expired") + require.NoError(t, err) + _, err = s.GetOAuth2AuthorizeRequestByID(ctx, "pending-fresh") + require.NoError(t, err) +} + +func TestRevokeOAuth2RefreshTokensByMode(t *testing.T) { + s := setupOAuth2TestStore(t) + ctx := context.Background() + sessionTok := makeRefreshToken("s1", "f1", "c", "h-session") + sessionTok.BfMode = "session" + vkTok := makeRefreshToken("v1", "f2", "c", "h-vk") // BfMode "vk" from helper default + require.NoError(t, s.DB().Create(sessionTok).Error) + require.NoError(t, s.DB().Create(vkTok).Error) + + require.NoError(t, s.RevokeOAuth2RefreshTokensByMode(ctx, "session")) + + // Only session-mode tokens revoked; vk-mode untouched. + _, err := s.GetOAuth2RefreshTokenByHash(ctx, "h-session") + assert.ErrorIs(t, err, ErrNotFound) + _, err = s.GetOAuth2RefreshTokenByHash(ctx, "h-vk") + require.NoError(t, err) +} + +func TestListOAuth2Sessions_JoinsAndExcludesRevoked(t *testing.T) { + s := setupOAuth2TestStore(t) + ctx := context.Background() + + require.NoError(t, s.DB().Create(&tables.TableOAuth2Client{ + ID: "crow", ClientID: "client-1", ClientName: "Test Client", + RedirectURIs: []string{"http://127.0.0.1/cb"}, GrantTypes: []string{"authorization_code"}, CreatedAt: time.Now(), + }).Error) + require.NoError(t, s.DB().Create(&tables.TableVirtualKey{ID: "vk-1", Name: "Alpha VK", Value: *schemas.NewSecretVar("sk-bf-alpha")}).Error) + + vkTok := makeRefreshToken("rt-vk", "f1", "client-1", "h-vk") + vkTok.BfSub = "vk-1" // joins to governance_virtual_keys.id + sessTok := makeRefreshToken("rt-sess", "f2", "client-1", "h-sess") + sessTok.BfMode = "session" + sessTok.BfSub = "sess-xyz" + revokedAt := time.Now() + deadTok := makeRefreshToken("rt-dead", "f3", "client-1", "h-dead") + deadTok.RevokedAt = &revokedAt + require.NoError(t, s.DB().Create(vkTok).Error) + require.NoError(t, s.DB().Create(sessTok).Error) + require.NoError(t, s.DB().Create(deadTok).Error) + + rows, total, err := s.ListOAuth2Sessions(ctx, OAuth2SessionsQueryParams{}) + require.NoError(t, err) + require.Len(t, rows, 2, "revoked grants are excluded") + require.Equal(t, int64(2), total, "total count excludes revoked grants") + + byID := map[string]OAuth2SessionRow{} + for _, r := range rows { + byID[r.ID] = r + } + assert.Equal(t, "Test Client", byID["rt-vk"].ClientName) + assert.Equal(t, "Alpha VK", byID["rt-vk"].BfSubDisplay, "vk mode resolves the VK name") + assert.Empty(t, byID["rt-sess"].BfSubDisplay, "session mode has no display name") + + // Round-trip the per-id load + revoke gate used by the management API. + got, err := s.GetOAuth2SessionByID(ctx, "rt-vk") + require.NoError(t, err) + assert.Equal(t, "vk", got.BfMode) + require.NoError(t, s.RevokeOAuth2Session(ctx, "rt-vk")) + _, err = s.GetOAuth2SessionByID(ctx, "rt-vk") + assert.ErrorIs(t, err, ErrNotFound) + // Revoking an already-revoked grant reports not-found. + assert.ErrorIs(t, s.RevokeOAuth2Session(ctx, "rt-dead"), ErrNotFound) +} + +// TestListOAuth2Sessions_FilterAndPaginate pins the DB-side filtering + +// pagination: ordering (created_at DESC), limit/offset paging, the total count +// (independent of the page slice), the bf_mode filter, and case-insensitive +// search across the joined client name, the joined VK display name, and the +// bound identity (bf_sub). +func TestListOAuth2Sessions_FilterAndPaginate(t *testing.T) { + s := setupOAuth2TestStore(t) + ctx := context.Background() + + require.NoError(t, s.DB().Create(&tables.TableOAuth2Client{ + ID: "c1", ClientID: "client-1", ClientName: "Acme Server", + RedirectURIs: []string{"http://127.0.0.1/cb"}, GrantTypes: []string{"authorization_code"}, CreatedAt: time.Now(), + }).Error) + require.NoError(t, s.DB().Create(&tables.TableOAuth2Client{ + ID: "c2", ClientID: "client-2", ClientName: "Beta Server", + RedirectURIs: []string{"http://127.0.0.1/cb"}, GrantTypes: []string{"authorization_code"}, CreatedAt: time.Now(), + }).Error) + require.NoError(t, s.DB().Create(&tables.TableVirtualKey{ID: "vk-1", Name: "Alpha VK", Value: *schemas.NewSecretVar("sk-bf-alpha")}).Error) + + base := time.Now() + rtA := makeRefreshToken("rt-a", "fa", "client-1", "h-a") // vk mode, bf_sub vk-1 → display "Alpha VK" + rtA.CreatedAt = base.Add(-3 * time.Minute) + rtB := makeRefreshToken("rt-b", "fb", "client-1", "h-b") + rtB.BfMode, rtB.BfSub = "user", "user@acme.com" + rtB.CreatedAt = base.Add(-2 * time.Minute) + rtC := makeRefreshToken("rt-c", "fc", "client-2", "h-c") + rtC.BfMode, rtC.BfSub = "session", "sess-xyz" + rtC.CreatedAt = base.Add(-1 * time.Minute) + require.NoError(t, s.DB().Create(rtA).Error) + require.NoError(t, s.DB().Create(rtB).Error) + require.NoError(t, s.DB().Create(rtC).Error) + + ids := func(rows []OAuth2SessionRow) []string { + out := make([]string, len(rows)) + for i, r := range rows { + out[i] = r.ID + } + return out + } + + // Page 1 (newest first), limit 2 — total reflects all matches, not the page. + rows, total, err := s.ListOAuth2Sessions(ctx, OAuth2SessionsQueryParams{Limit: 2}) + require.NoError(t, err) + assert.Equal(t, int64(3), total) + assert.Equal(t, []string{"rt-c", "rt-b"}, ids(rows), "ordered created_at DESC") + + // Page 2. + rows, total, err = s.ListOAuth2Sessions(ctx, OAuth2SessionsQueryParams{Limit: 2, Offset: 2}) + require.NoError(t, err) + assert.Equal(t, int64(3), total) + assert.Equal(t, []string{"rt-a"}, ids(rows)) + + // bf_mode filter. + rows, total, err = s.ListOAuth2Sessions(ctx, OAuth2SessionsQueryParams{Modes: []string{"user"}}) + require.NoError(t, err) + assert.Equal(t, int64(1), total) + assert.Equal(t, []string{"rt-b"}, ids(rows)) + + // Search matches the joined VK display name. + rows, total, err = s.ListOAuth2Sessions(ctx, OAuth2SessionsQueryParams{Search: "alpha"}) + require.NoError(t, err) + assert.Equal(t, int64(1), total) + assert.Equal(t, []string{"rt-a"}, ids(rows)) + + // Search matches the joined client name. + rows, total, err = s.ListOAuth2Sessions(ctx, OAuth2SessionsQueryParams{Search: "beta"}) + require.NoError(t, err) + assert.Equal(t, int64(1), total) + assert.Equal(t, []string{"rt-c"}, ids(rows)) + + // Search matches the bound identity (bf_sub). + rows, total, err = s.ListOAuth2Sessions(ctx, OAuth2SessionsQueryParams{Search: "user@acme"}) + require.NoError(t, err) + assert.Equal(t, int64(1), total) + assert.Equal(t, []string{"rt-b"}, ids(rows)) +} + +// TestListOAuth2Sessions_StableTiebreakerSameTimestamp pins the secondary id sort: +// when grants share an identical created_at, ordering by created_at alone is +// nondeterministic, so offset paging could repeat or skip rows. The id tiebreaker +// makes every page deterministic — here the two pages partition all four rows +// (ordered id DESC) with no duplicates and no gaps. +func TestListOAuth2Sessions_StableTiebreakerSameTimestamp(t *testing.T) { + s := setupOAuth2TestStore(t) + ctx := context.Background() + + ids := func(rows []OAuth2SessionRow) []string { + out := make([]string, len(rows)) + for i, r := range rows { + out[i] = r.ID + } + return out + } + + // All four grants share the same created_at, so only the id tiebreaker can + // give a stable order. + ts := time.Now() + for _, id := range []string{"rt-1", "rt-2", "rt-3", "rt-4"} { + rt := makeRefreshToken(id, "fam-"+id, "client-x", "hash-"+id) + rt.CreatedAt = ts + require.NoError(t, s.DB().Create(rt).Error) + } + + // Page 1 and Page 2 (limit 2 each) must partition all rows in id-DESC order. + page1, total, err := s.ListOAuth2Sessions(ctx, OAuth2SessionsQueryParams{Limit: 2}) + require.NoError(t, err) + assert.Equal(t, int64(4), total) + assert.Equal(t, []string{"rt-4", "rt-3"}, ids(page1), "page 1 ordered by id DESC on tied timestamps") + + page2, total, err := s.ListOAuth2Sessions(ctx, OAuth2SessionsQueryParams{Limit: 2, Offset: 2}) + require.NoError(t, err) + assert.Equal(t, int64(4), total) + assert.Equal(t, []string{"rt-2", "rt-1"}, ids(page2), "page 2 continues without overlap or gap") +} + +// TestSweepConvergence_TokenSweepThenClientSweep pins the documented ordering: a +// client whose only tokens are revoked is collected only after the token sweep +// removes those aged rows, leaving the client backing zero tokens. +func TestSweepConvergence_TokenSweepThenClientSweep(t *testing.T) { + s := setupOAuth2TestStore(t) + ctx := context.Background() + old := time.Now().Add(-2 * time.Hour) + + require.NoError(t, s.DB().Create(&tables.TableOAuth2Client{ + ID: "c", ClientID: "revoked-only", RedirectURIs: []string{"http://127.0.0.1/cb"}, + GrantTypes: []string{"authorization_code"}, CreatedAt: old, + }).Error) + revokedAt := old + tok := makeRefreshToken("rt", "fam", "revoked-only", "h") + tok.RevokedAt = &revokedAt + require.NoError(t, s.DB().Create(tok).Error) + + // Before the token sweep, the client still backs a (revoked) token row → kept. + deleted, err := s.SweepOrphanedOAuth2Clients(ctx, time.Hour) + require.NoError(t, err) + assert.Equal(t, int64(0), deleted) + + // Token sweep removes the aged revoked row, then the client is collectible. + _, err = s.SweepOAuth2RefreshTokens(ctx, time.Hour) + require.NoError(t, err) + deleted, err = s.SweepOrphanedOAuth2Clients(ctx, time.Hour) + require.NoError(t, err) + assert.Equal(t, int64(1), deleted) + _, err = s.GetOAuth2ClientByClientID(ctx, "revoked-only") + assert.ErrorIs(t, err, ErrNotFound) +} diff --git a/framework/configstore/rdb_test.go b/framework/configstore/rdb_test.go index 9d876e710d7..a385c261e3c 100644 --- a/framework/configstore/rdb_test.go +++ b/framework/configstore/rdb_test.go @@ -55,6 +55,7 @@ func setupRDBTestStore(t *testing.T) *RDBConfigStore { &tables.TableOauthUserToken{}, &tables.TableMCPPerUserHeaderCredential{}, &tables.TableMCPPerUserHeaderFlow{}, + &tables.TableOAuth2RefreshToken{}, ) require.NoError(t, err, "Failed to migrate test database") @@ -87,6 +88,36 @@ func testComplexityAnalyzerConfig() *ComplexityAnalyzerConfig { } } +func TestRDBConfigStore_UpsertModelPricesSyncsIsDeprecated(t *testing.T) { + store := setupRDBTestStore(t) + require.NoError(t, store.DB().AutoMigrate(&tables.TableModelPricing{})) + ctx := context.Background() + + require.NoError(t, store.UpsertModelPrices(ctx, &tables.TableModelPricing{ + Model: "deprecated-model", + Provider: "openai", + Mode: "chat", + IsDeprecated: true, + })) + + prices, err := store.GetModelPrices(ctx) + require.NoError(t, err) + require.Len(t, prices, 1) + assert.True(t, prices[0].IsDeprecated) + + require.NoError(t, store.UpsertModelPrices(ctx, &tables.TableModelPricing{ + Model: "deprecated-model", + Provider: "openai", + Mode: "chat", + IsDeprecated: false, + })) + + prices, err = store.GetModelPrices(ctx) + require.NoError(t, err) + require.Len(t, prices, 1) + assert.False(t, prices[0].IsDeprecated) +} + func TestRDBConfigStore_ComplexityAnalyzerConfigRoundTrip(t *testing.T) { store := setupRDBTestStore(t) ctx := context.Background() @@ -835,7 +866,7 @@ func TestCreateVirtualKey(t *testing.T) { vk := &tables.TableVirtualKey{ ID: "vk-test", Name: "Test Virtual Key", - Value: "vk-test-value-123", + Value: *schemas.NewSecretVar("vk-test-value-123"), IsActive: schemas.Ptr(true), } @@ -846,7 +877,7 @@ func TestCreateVirtualKey(t *testing.T) { require.NoError(t, err) assert.Equal(t, "vk-test", result.ID) assert.Equal(t, "Test Virtual Key", result.Name) - assert.Equal(t, "vk-test-value-123", result.Value) + assert.Equal(t, "vk-test-value-123", result.Value.Val) assert.True(t, result.IsActiveValue()) } @@ -880,7 +911,7 @@ func TestCreateVirtualKey_WithBudgetAndRateLimit(t *testing.T) { vk := &tables.TableVirtualKey{ ID: vkID, Name: "VK With References", - Value: "vk-refs-value", + Value: *schemas.NewSecretVar("vk-refs-value"), IsActive: schemas.Ptr(true), RateLimitID: &rateLimitID, } @@ -908,7 +939,7 @@ func TestCreateVirtualKey_DuplicateName(t *testing.T) { vk1 := &tables.TableVirtualKey{ ID: "vk-1", Name: "Same Name", - Value: "vk-value-1", + Value: *schemas.NewSecretVar("vk-value-1"), IsActive: schemas.Ptr(true), } err := store.CreateVirtualKey(ctx, vk1) @@ -917,7 +948,7 @@ func TestCreateVirtualKey_DuplicateName(t *testing.T) { vk2 := &tables.TableVirtualKey{ ID: "vk-2", Name: "Same Name", // Duplicate name - Value: "vk-value-2", + Value: *schemas.NewSecretVar("vk-value-2"), IsActive: schemas.Ptr(true), } err = store.CreateVirtualKey(ctx, vk2) @@ -931,7 +962,7 @@ func TestGetVirtualKeyByValue(t *testing.T) { vk := &tables.TableVirtualKey{ ID: "vk-lookup", Name: "Lookup Key", - Value: "vk-unique-value-xyz", + Value: *schemas.NewSecretVar("vk-unique-value-xyz"), IsActive: schemas.Ptr(true), } err := store.CreateVirtualKey(ctx, vk) @@ -949,7 +980,7 @@ func TestUpdateVirtualKey(t *testing.T) { vk := &tables.TableVirtualKey{ ID: "vk-update", Name: "Original Name", - Value: "vk-update-value", + Value: *schemas.NewSecretVar("vk-update-value"), IsActive: schemas.Ptr(true), } err := store.CreateVirtualKey(ctx, vk) @@ -974,7 +1005,7 @@ func TestDeleteVirtualKey(t *testing.T) { vk := &tables.TableVirtualKey{ ID: "vk-delete", Name: "Delete Me", - Value: "vk-delete-value", + Value: *schemas.NewSecretVar("vk-delete-value"), IsActive: schemas.Ptr(true), } err := store.CreateVirtualKey(ctx, vk) @@ -987,6 +1018,40 @@ func TestDeleteVirtualKey(t *testing.T) { assert.Error(t, err, "Should not find deleted virtual key") } +func TestDeleteVirtualKey_RevokesInboundVKGrants(t *testing.T) { + store := setupRDBTestStore(t) + ctx := context.Background() + + vk := &tables.TableVirtualKey{ + ID: "vk-grant", + Name: "Grant VK", + Value: *schemas.NewSecretVar("vk-grant-value"), + IsActive: schemas.Ptr(true), + } + require.NoError(t, store.CreateVirtualKey(ctx, vk)) + + // An active vk-mode inbound grant bound to this VK (vk-mode rows key bf_sub + // to the VK id). + rt := &tables.TableOAuth2RefreshToken{ + ID: "rt-vk-grant", + TokenHash: "hash-vk-grant", + FamilyID: "fam-vk-grant", + ClientID: "client-1", + BfMode: string(schemas.MCPAuthModeVK), + BfSub: vk.ID, + Scope: "mcp", + Resource: "https://example.test/mcp", + CreatedAt: time.Now(), + } + require.NoError(t, store.DB().WithContext(ctx).Create(rt).Error) + + require.NoError(t, store.DeleteVirtualKey(ctx, vk.ID)) + + var got tables.TableOAuth2RefreshToken + require.NoError(t, store.DB().WithContext(ctx).First(&got, "id = ?", "rt-vk-grant").Error) + assert.NotNil(t, got.RevokedAt, "vk-mode grant should be revoked when its VK is deleted") +} + func TestDeleteVirtualKey_CleansUpScopedModelConfigs(t *testing.T) { store := setupRDBTestStore(t) ctx := context.Background() @@ -994,7 +1059,7 @@ func TestDeleteVirtualKey_CleansUpScopedModelConfigs(t *testing.T) { vk := &tables.TableVirtualKey{ ID: "vk-scoped", Name: "Scoped VK", - Value: "vk-scoped-value", + Value: *schemas.NewSecretVar("vk-scoped-value"), IsActive: schemas.Ptr(true), } require.NoError(t, store.CreateVirtualKey(ctx, vk)) @@ -1048,7 +1113,7 @@ func TestDeleteVirtualKey_CleansUpMultiBudgetScopedModelConfigs(t *testing.T) { vk := &tables.TableVirtualKey{ ID: "vk-multibudget", Name: "MultiBudget VK", - Value: "vk-multibudget-value", + Value: *schemas.NewSecretVar("vk-multibudget-value"), IsActive: schemas.Ptr(true), } require.NoError(t, store.CreateVirtualKey(ctx, vk)) @@ -1171,7 +1236,7 @@ func TestCreateVirtualKeyProviderConfig(t *testing.T) { vk := &tables.TableVirtualKey{ ID: "vk-for-pc", Name: "VK For Provider Config", - Value: "vk-pc-value", + Value: *schemas.NewSecretVar("vk-pc-value"), IsActive: schemas.Ptr(true), } err := store.CreateVirtualKey(ctx, vk) @@ -1214,7 +1279,7 @@ func TestCreateVirtualKeyProviderConfig_WithKeys(t *testing.T) { vk := &tables.TableVirtualKey{ ID: "vk-with-keys", Name: "VK With Keys", - Value: "vk-keys-value", + Value: *schemas.NewSecretVar("vk-keys-value"), IsActive: schemas.Ptr(true), } err = store.CreateVirtualKey(ctx, vk) @@ -1254,7 +1319,7 @@ func TestCreateVirtualKeyProviderConfig_UnresolvedKeys(t *testing.T) { vk := &tables.TableVirtualKey{ ID: "vk-unresolved", Name: "VK Unresolved", - Value: "vk-unresolved-value", + Value: *schemas.NewSecretVar("vk-unresolved-value"), IsActive: schemas.Ptr(true), } err := store.CreateVirtualKey(ctx, vk) @@ -1296,7 +1361,7 @@ func TestUpdateProvider_RemovesStaleVirtualKeyProviderConfigKeyAssociations(t *t vk := &tables.TableVirtualKey{ ID: "vk-update-provider-cleanup", Name: "VK Update Provider Cleanup", - Value: "vk-update-provider-cleanup-value", + Value: *schemas.NewSecretVar("vk-update-provider-cleanup-value"), IsActive: schemas.Ptr(true), } err = store.CreateVirtualKey(ctx, vk) @@ -1345,7 +1410,7 @@ func TestDeleteProvider_RemovesVirtualKeyProviderConfigs(t *testing.T) { vk := &tables.TableVirtualKey{ ID: "vk-delete-provider-cleanup", Name: "VK Delete Provider Cleanup", - Value: "vk-delete-provider-cleanup-value", + Value: *schemas.NewSecretVar("vk-delete-provider-cleanup-value"), IsActive: schemas.Ptr(true), } err = store.CreateVirtualKey(ctx, vk) @@ -1741,7 +1806,7 @@ func TestFullVirtualKeyFlow(t *testing.T) { vk := &tables.TableVirtualKey{ ID: integrationVKID, Name: "Integration Virtual Key", - Value: "vk-integration-xyz", + Value: *schemas.NewSecretVar("vk-integration-xyz"), IsActive: schemas.Ptr(true), RateLimitID: &rateLimitID, } @@ -1792,7 +1857,7 @@ func TestGetVirtualKeysUsesInternalPagination(t *testing.T) { vk := &tables.TableVirtualKey{ ID: fmt.Sprintf("vk-page-%04d", i), Name: fmt.Sprintf("Virtual Key %04d", i), - Value: fmt.Sprintf("vk-value-%04d", i), + Value: *schemas.NewSecretVar(fmt.Sprintf("vk-value-%04d", i)), IsActive: schemas.Ptr(true), CreatedAt: createdAt, UpdatedAt: createdAt, @@ -2211,3 +2276,40 @@ func TestUpsertModelPricesBatch_SQLite(t *testing.T) { require.NotNil(t, updated.InputCostPerToken) assert.InDelta(t, 0.000005, *updated.InputCostPerToken, 1e-9) } + +func TestUpsertModelParametersBatch_SQLite(t *testing.T) { + s := setupRDBTestStore(t) + require.NoError(t, s.DB().AutoMigrate(&tables.TableModelParameters{})) + + ctx := context.Background() + params := []tables.TableModelParameters{ + {Model: "model-a", Data: `{"max_output_tokens":100}`}, + {Model: "model-b", Data: `{"max_output_tokens":200}`}, + {Model: "model-c", Data: `{"max_output_tokens":300}`}, + } + + require.NoError(t, s.UpsertModelParametersBatch(ctx, params)) + + got, err := s.GetModelParameters(ctx) + require.NoError(t, err) + assert.Len(t, got, 3) + + params[1].Data = `{"max_output_tokens":250}` + require.NoError(t, s.UpsertModelParametersBatch(ctx, params)) + + updated, err := s.GetModelParametersByModel(ctx, "model-b") + require.NoError(t, err) + assert.Equal(t, `{"max_output_tokens":250}`, updated.Data) + + require.NoError(t, s.UpsertModelParametersBatch(ctx, []tables.TableModelParameters{ + {Model: "model-b", Data: `{"max_output_tokens":260}`}, + {Model: "model-b", Data: `{"max_output_tokens":270}`}, + })) + updated, err = s.GetModelParametersByModel(ctx, "model-b") + require.NoError(t, err) + assert.Equal(t, `{"max_output_tokens":270}`, updated.Data) + + got, err = s.GetModelParameters(ctx) + require.NoError(t, err) + assert.Len(t, got, 3) +} diff --git a/framework/configstore/store.go b/framework/configstore/store.go index 9126113accd..d6cb6f69c6f 100644 --- a/framework/configstore/store.go +++ b/framework/configstore/store.go @@ -64,9 +64,27 @@ type RoutingRulesQueryParams struct { // MCPClientsQueryParams holds pagination, filtering, and search parameters for MCP client queries. type MCPClientsQueryParams struct { - Limit int - Offset int - Search string + Limit int + Offset int + Search string // matches name (case-insensitive) + ClientID string // exact client_id match + ConnectionTypes []string // exact connection_type filter(s), OR semantics (http | sse | stdio) + AuthTypes []string // exact auth_type filter(s), OR semantics (none | headers | oauth | per_user_oauth | per_user_headers) + IsCodeModeClient *bool // nil = no filter; true/false = filter on is_code_mode_client + Disabled *bool // nil = no filter; true/false = filter on disabled + + // Runtime connection-state filter. State is not persisted, so the caller + // resolves the set of currently-connected client_ids from the engine and + // passes it here. StateInclude nil = no filter; true = client_id IN set + // (connected); false = client_id NOT IN set (disconnected). + StateClientIDs []string + StateInclude *bool + + // Virtual-key access filter (OR semantics within the group). When both are + // set, a client matches if it is open to all VKs OR explicitly assigned to + // one of VirtualKeyIDs. + OnlyAllVirtualKeys bool // include clients with allow_on_all_virtual_keys=true + VirtualKeyIDs []string // include clients explicitly assigned to any of these VK IDs } // MCPLibraryQueryParams holds pagination, filtering, search, and sort @@ -126,6 +144,12 @@ type MCPSessionsFilterParams struct { Statuses []string AuthModes []string // matched against auth_mode (tokens, credentials) or flow_mode (sessions, flows) MCPClientIDs []string + // Identity exact-matches a single resolved identity value against any of + // the row's identity columns (user_id, virtual_key_id, session_id). Unlike + // Search it is not a substring match — it pins the list to exactly one + // user, virtual key, or session. Typically paired with AuthModes to scope + // to that identity's rows for a known mode. + Identity string // MatchedUserIDs is an optional set of user_ids that should be treated // as a positive search hit alongside Search. Callers that maintain a // user directory (display names, emails) resolve the search string @@ -137,6 +161,18 @@ type MCPSessionsFilterParams struct { MatchedUserIDs []string } +// OAuth2SessionsQueryParams holds the filters + pagination for the OAuth2 +// grants list (Connected Clients UI). Search is a case-insensitive substring +// matched against the client name/id, the bound identity (bf_sub), and the +// joined virtual key name. Modes filters on bf_mode (user/vk/session); an +// empty slice matches all. Limit/Offset paginate the filtered result in SQL. +type OAuth2SessionsQueryParams struct { + Search string + Modes []string + Limit int + Offset int +} + // PricingOverrideFilters holds the filters for pricing overrides. type PricingOverrideFilters struct { ScopeKind *string @@ -389,6 +425,7 @@ type ConfigStore interface { GetModelParameters(ctx context.Context) ([]tables.TableModelParameters, error) GetModelParametersByModel(ctx context.Context, model string) (*tables.TableModelParameters, error) UpsertModelParameters(ctx context.Context, params *tables.TableModelParameters, tx ...*gorm.DB) error + UpsertModelParametersBatch(ctx context.Context, params []tables.TableModelParameters, tx ...*gorm.DB) error // Key management GetKeysByIDs(ctx context.Context, ids []string) ([]tables.TableKey, error) @@ -655,6 +692,57 @@ type ConfigStore interface { // pool, whose connections carry no cached plans. SQLite is a no-op. RefreshConnectionPool(ctx context.Context) error + // GetOAuth2SigningKey returns the signing key, creating and persisting one + // on first call. Always returns a usable key — never nil on a nil error. + GetOAuth2SigningKey(ctx context.Context) (*tables.OAuth2SigningKey, error) + + // OAuth2 clients (DCR) + CreateOAuth2Client(ctx context.Context, client *tables.TableOAuth2Client) error + GetOAuth2ClientByClientID(ctx context.Context, clientID string) (*tables.TableOAuth2Client, error) + + // OAuth2 authorize requests + CreateOAuth2AuthorizeRequest(ctx context.Context, req *tables.TableOAuth2AuthorizeRequest) error + GetOAuth2AuthorizeRequestByID(ctx context.Context, id string) (*tables.TableOAuth2AuthorizeRequest, error) + GetOAuth2AuthorizeRequestByCodeHash(ctx context.Context, codeHash string) (*tables.TableOAuth2AuthorizeRequest, error) + // ConsentOAuth2AuthorizeRequest atomically transitions a still-pending request + // to consented (recording the code hash and resolved identity) — returns + // ErrNotFound when no longer pending, so concurrent double-consent can't + // overwrite an already-minted code. + ConsentOAuth2AuthorizeRequest(ctx context.Context, req *tables.TableOAuth2AuthorizeRequest) error + SweepExpiredOAuth2AuthorizeRequests(ctx context.Context) error + + // OAuth2 refresh tokens + GetOAuth2RefreshTokenByHash(ctx context.Context, hash string) (*tables.TableOAuth2RefreshToken, error) + // GetOAuth2RefreshTokenByHashAny returns the row including revoked tokens, + // used to detect token reuse attacks and trigger family revocation. + GetOAuth2RefreshTokenByHashAny(ctx context.Context, hash string) (*tables.TableOAuth2RefreshToken, error) + // ConsumeOAuth2AuthorizeRequest atomically marks the authorize request as + // code_issued and creates the refresh token — if either fails the client can retry. + ConsumeOAuth2AuthorizeRequest(ctx context.Context, requestID string, rt *tables.TableOAuth2RefreshToken) error + // RotateOAuth2RefreshToken atomically revokes the old token and creates the + // new one — if either fails the old token stays active and the client can retry. + RotateOAuth2RefreshToken(ctx context.Context, oldID string, newRT *tables.TableOAuth2RefreshToken) error + // RevokeOAuth2RefreshTokensByFamilyID revokes all active tokens in a family + // when a stolen-token reuse is detected (RFC 9700 §2.2.2). + RevokeOAuth2RefreshTokensByFamilyID(ctx context.Context, familyID string) error + // RevokeOAuth2RefreshTokensByMode revokes all active tokens for a given mode. + RevokeOAuth2RefreshTokensByMode(ctx context.Context, bfMode string) error + // SweepOAuth2RefreshTokens deletes revoked tokens older than the given duration. + SweepOAuth2RefreshTokens(ctx context.Context, revokedOlderThan time.Duration) (int64, error) + // SweepOrphanedOAuth2Clients deletes registered clients that back no refresh + // token and were registered before the grace cutoff. Run after the refresh + // token sweep so clients are not collected while their tokens are still + // retained for reuse detection. + SweepOrphanedOAuth2Clients(ctx context.Context, registeredOlderThan time.Duration) (int64, error) + // ListOAuth2Sessions returns a page of active downstream grants for the + // Connected Clients UI, plus the total count matching the filters (before + // the limit/offset are applied). Filtering and pagination are pushed to SQL. + ListOAuth2Sessions(ctx context.Context, params OAuth2SessionsQueryParams) ([]OAuth2SessionRow, int64, error) + // GetOAuth2SessionByID returns a single active grant row for permission checks. + GetOAuth2SessionByID(ctx context.Context, id string) (*tables.TableOAuth2RefreshToken, error) + // RevokeOAuth2Session revokes a specific downstream grant by refresh token ID. + RevokeOAuth2Session(ctx context.Context, id string) error + // Cleanup Close(ctx context.Context) error } diff --git a/framework/configstore/tables/clientconfig.go b/framework/configstore/tables/clientconfig.go index 642100f5460..3f1a6cf6ac0 100644 --- a/framework/configstore/tables/clientconfig.go +++ b/framework/configstore/tables/clientconfig.go @@ -49,6 +49,16 @@ type TableClientConfig struct { CompatShouldDropParams bool `gorm:"column:compat_should_drop_params;default:false" json:"-"` CompatShouldConvertParams bool `gorm:"column:compat_should_convert_params;default:false" json:"-"` + // MCPServerAuthMode controls how /mcp authenticates inbound clients. + // Stored as a plain varchar column so it can be read without JSON parsing. + MCPServerAuthMode MCPServerAuthMode `gorm:"column:mcp_server_auth_mode;type:varchar(20);not null;default:'headers'" json:"mcp_server_auth_mode"` + // OAuth2ServerConfigJSON holds the OAuth2 AS-specific settings (IssuerURL, + // AuthCodeTTL, AccessTokenTTL) as a JSON blob. Only relevant when + // MCPServerAuthMode is both or oauth. Deserialized into OAuth2ServerConfig + // by AfterFind. The explicit column name avoids GORM deriving the leading + // acronym as "o_auth2_..." from the field name. + OAuth2ServerConfigJSON string `gorm:"column:oauth2_server_config_json;type:text" json:"-"` + // Config hash is used to detect the changes synced from config.json file // Every time we sync the config.json file, we will update the config hash ConfigHash string `gorm:"type:varchar(255);null" json:"config_hash"` @@ -57,14 +67,15 @@ type TableClientConfig struct { UpdatedAt time.Time `gorm:"index;not null" json:"updated_at"` // Virtual fields for runtime use (not stored in DB) - PrometheusLabels []string `gorm:"-" json:"prometheus_labels"` - AllowedOrigins []string `gorm:"-" json:"allowed_origins,omitempty"` - AllowedHeaders []string `gorm:"-" json:"allowed_headers,omitempty"` - RequiredHeaders []string `gorm:"-" json:"required_headers,omitempty"` - LoggingHeaders []string `gorm:"-" json:"logging_headers,omitempty"` - WhitelistedRoutes []string `gorm:"-" json:"whitelisted_routes,omitempty"` - HeaderFilterConfig *GlobalHeaderFilterConfig `gorm:"-" json:"header_filter_config,omitempty"` - Metadata map[string]any `gorm:"-" json:"metadata,omitempty"` + PrometheusLabels []string `gorm:"-" json:"prometheus_labels"` + AllowedOrigins []string `gorm:"-" json:"allowed_origins,omitempty"` + AllowedHeaders []string `gorm:"-" json:"allowed_headers,omitempty"` + RequiredHeaders []string `gorm:"-" json:"required_headers,omitempty"` + LoggingHeaders []string `gorm:"-" json:"logging_headers,omitempty"` + WhitelistedRoutes []string `gorm:"-" json:"whitelisted_routes,omitempty"` + HeaderFilterConfig *GlobalHeaderFilterConfig `gorm:"-" json:"header_filter_config,omitempty"` + Metadata map[string]any `gorm:"-" json:"metadata,omitempty"` + OAuth2ServerConfig *OAuth2ServerConfig `gorm:"-" json:"oauth2_server_config,omitempty"` } // TableName sets the table name for each model @@ -153,6 +164,16 @@ func (cc *TableClientConfig) BeforeSave(tx *gorm.DB) error { cc.MetadataJSON = string(data) } + if cc.OAuth2ServerConfig != nil { + data, err := json.Marshal(cc.OAuth2ServerConfig) + if err != nil { + return err + } + cc.OAuth2ServerConfigJSON = string(data) + } else { + cc.OAuth2ServerConfigJSON = "" + } + return nil } @@ -212,5 +233,15 @@ func (cc *TableClientConfig) AfterFind(tx *gorm.DB) error { cc.Metadata = nil } + if cc.OAuth2ServerConfigJSON != "" { + var authCfg OAuth2ServerConfig + if err := json.Unmarshal([]byte(cc.OAuth2ServerConfigJSON), &authCfg); err != nil { + return err + } + cc.OAuth2ServerConfig = &authCfg + } else { + cc.OAuth2ServerConfig = nil + } + return nil } diff --git a/framework/configstore/tables/encryption_test.go b/framework/configstore/tables/encryption_test.go index b93b298685d..d243d3a5e16 100644 --- a/framework/configstore/tables/encryption_test.go +++ b/framework/configstore/tables/encryption_test.go @@ -408,7 +408,7 @@ func TestTableVirtualKey_EncryptDecrypt(t *testing.T) { vk := &TableVirtualKey{ ID: "vk-1", Name: "test-vk", - Value: "vk-secret-value-xyz", + Value: *schemas.NewSecretVar("vk-secret-value-xyz"), IsActive: bifrost.Ptr(true), } @@ -424,7 +424,7 @@ func TestTableVirtualKey_EncryptDecrypt(t *testing.T) { var found TableVirtualKey require.NoError(t, db.First(&found, "id = ?", "vk-1").Error) - assert.Equal(t, "vk-secret-value-xyz", found.Value) + assert.Equal(t, "vk-secret-value-xyz", found.Value.GetValue()) assert.Equal(t, expectedHash, found.ValueHash) } @@ -434,7 +434,7 @@ func TestTableVirtualKey_HashComputedBeforeEncryption(t *testing.T) { vk := &TableVirtualKey{ ID: "vk-hash", Name: "hash-test", - Value: "plaintext-value", + Value: *schemas.NewSecretVar("plaintext-value"), IsActive: bifrost.Ptr(true), } @@ -892,21 +892,21 @@ func TestTableVirtualKey_UpdatePreservesDecryption(t *testing.T) { vk := &TableVirtualKey{ ID: "vk-update", Name: "update-vk", - Value: "original-vk-value", + Value: *schemas.NewSecretVar("original-vk-value"), IsActive: bifrost.Ptr(true), } require.NoError(t, db.Create(vk).Error) var found TableVirtualKey require.NoError(t, db.First(&found, "id = ?", "vk-update").Error) - assert.Equal(t, "original-vk-value", found.Value) + assert.Equal(t, "original-vk-value", found.Value.GetValue()) - found.Value = "updated-vk-value" + found.Value = *schemas.NewSecretVar("updated-vk-value") require.NoError(t, db.Save(&found).Error) var found2 TableVirtualKey require.NoError(t, db.First(&found2, "id = ?", "vk-update").Error) - assert.Equal(t, "updated-vk-value", found2.Value) + assert.Equal(t, "updated-vk-value", found2.Value.GetValue()) raw := rawRow(t, db, "governance_virtual_keys", "vk-update") assert.Equal(t, "encrypted", raw["encryption_status"]) @@ -1315,7 +1315,7 @@ func TestTableVirtualKey_EncryptionDisabled_StoresPlaintext(t *testing.T) { vk := &TableVirtualKey{ ID: "vk-dis-1", Name: "disabled-vk", - Value: "vk-plaintext-value", + Value: *schemas.NewSecretVar("vk-plaintext-value"), IsActive: bifrost.Ptr(true), } @@ -1331,7 +1331,7 @@ func TestTableVirtualKey_EncryptionDisabled_StoresPlaintext(t *testing.T) { // GORM read should return same plaintext var found TableVirtualKey require.NoError(t, db.Where("id = ?", "vk-dis-1").First(&found).Error) - assert.Equal(t, "vk-plaintext-value", found.Value) + assert.Equal(t, "vk-plaintext-value", found.Value.GetValue()) } func TestSessionsTable_EncryptionDisabled_StoresPlaintext(t *testing.T) { diff --git a/framework/configstore/tables/key.go b/framework/configstore/tables/key.go index 660cdaa88ce..3eec1f6c3db 100644 --- a/framework/configstore/tables/key.go +++ b/framework/configstore/tables/key.go @@ -13,18 +13,18 @@ import ( // TableKey represents an API key configuration in the database type TableKey struct { - ID uint `gorm:"primaryKey;autoIncrement" json:"id"` - Name string `gorm:"type:varchar(255);uniqueIndex:idx_key_name;not null" json:"name"` - ProviderID uint `gorm:"index;not null" json:"provider_id"` - Provider string `gorm:"index;type:varchar(50)" json:"provider"` // ModelProvider as string - KeyID string `gorm:"type:varchar(255);uniqueIndex:idx_key_id;not null" json:"key_id"` // UUID from schemas.Key + ID uint `gorm:"primaryKey;autoIncrement" json:"id"` + Name string `gorm:"type:varchar(255);uniqueIndex:idx_key_name;not null" json:"name"` + ProviderID uint `gorm:"index;not null" json:"provider_id"` + Provider string `gorm:"index;type:varchar(50)" json:"provider"` // ModelProvider as string + KeyID string `gorm:"type:varchar(255);uniqueIndex:idx_key_id;not null" json:"key_id"` // UUID from schemas.Key Value schemas.SecretVar `gorm:"type:text;not null" json:"value"` - ModelsJSON string `gorm:"type:text" json:"-"` // JSON serialized []string - BlacklistedModelsJSON string `gorm:"type:text" json:"-"` // JSON serialized []string - Weight *float64 `json:"weight"` - Enabled *bool `gorm:"default:true" json:"enabled,omitempty"` - CreatedAt time.Time `gorm:"index;not null" json:"created_at"` - UpdatedAt time.Time `gorm:"index;not null" json:"updated_at"` + ModelsJSON string `gorm:"type:text" json:"-"` // JSON serialized []string + BlacklistedModelsJSON string `gorm:"type:text" json:"-"` // JSON serialized []string + Weight *float64 `json:"weight"` + Enabled *bool `gorm:"default:true" json:"enabled,omitempty"` + CreatedAt time.Time `gorm:"index;not null" json:"created_at"` + UpdatedAt time.Time `gorm:"index;not null" json:"updated_at"` // Config hash is used to detect changes synced from config.json file ConfigHash string `gorm:"type:varchar(255);null" json:"config_hash"` @@ -37,7 +37,7 @@ type TableKey struct { AzureClientID *schemas.SecretVar `gorm:"type:text" json:"azure_client_id,omitempty"` AzureClientSecret *schemas.SecretVar `gorm:"type:text" json:"azure_client_secret,omitempty"` AzureTenantID *schemas.SecretVar `gorm:"type:text" json:"azure_tenant_id,omitempty"` - AzureScopesJSON *string `gorm:"column:azure_scopes;type:text" json:"-"` // JSON serialized []string + AzureScopesJSON *string `gorm:"column:azure_scopes;type:text" json:"-"` // JSON serialized []string // Vertex config fields (embedded) VertexProjectID *schemas.SecretVar `gorm:"type:text" json:"vertex_project_id,omitempty"` @@ -54,11 +54,20 @@ type TableKey struct { BedrockRoleARN *schemas.SecretVar `gorm:"type:text" json:"bedrock_role_arn,omitempty"` BedrockExternalID *schemas.SecretVar `gorm:"type:text" json:"bedrock_external_id,omitempty"` BedrockRoleSessionName *schemas.SecretVar `gorm:"type:text" json:"bedrock_role_session_name,omitempty"` - BedrockBatchS3ConfigJSON *string `gorm:"type:text" json:"-"` // JSON serialized schemas.BatchS3Config + BedrockBatchS3ConfigJSON *string `gorm:"type:text" json:"-"` // JSON serialized schemas.BatchS3Config + + // Bedrock Mantle config fields (embedded) + BedrockMantleAccessKey *schemas.SecretVar `gorm:"type:text" json:"bedrock_mantle_access_key,omitempty"` + BedrockMantleSecretKey *schemas.SecretVar `gorm:"type:text" json:"bedrock_mantle_secret_key,omitempty"` + BedrockMantleSessionToken *schemas.SecretVar `gorm:"type:text" json:"bedrock_mantle_session_token,omitempty"` + BedrockMantleRegion *schemas.SecretVar `gorm:"type:text" json:"bedrock_mantle_region,omitempty"` + BedrockMantleRoleARN *schemas.SecretVar `gorm:"type:text" json:"bedrock_mantle_role_arn,omitempty"` + BedrockMantleExternalID *schemas.SecretVar `gorm:"type:text" json:"bedrock_mantle_external_id,omitempty"` + BedrockMantleRoleSessionName *schemas.SecretVar `gorm:"type:text" json:"bedrock_mantle_role_session_name,omitempty"` // VLLM config fields (embedded) VLLMUrl *schemas.SecretVar `gorm:"type:text" json:"vllm_url,omitempty"` - VLLMModelName *string `gorm:"type:varchar(255)" json:"vllm_model_name,omitempty"` + VLLMModelName *string `gorm:"type:varchar(255)" json:"vllm_model_name,omitempty"` // Replicate config fields (embedded) ReplicateUseDeploymentsEndpoint *bool `gorm:"column:replicate_use_deployments_endpoint" json:"replicate_use_deployments_endpoint,omitempty"` @@ -78,16 +87,17 @@ type TableKey struct { EncryptionStatus string `gorm:"type:varchar(20);default:'plain_text'" json:"-"` // Virtual fields for runtime use (not stored in DB) - Models schemas.WhiteList `gorm:"-" json:"models"` // ["*"] allows all models; empty denies all (deny-by-default) - BlacklistedModels schemas.BlackList `gorm:"-" json:"blacklisted_models"` - Aliases schemas.KeyAliases `gorm:"-" json:"aliases,omitempty"` - AzureKeyConfig *schemas.AzureKeyConfig `gorm:"-" json:"azure_key_config,omitempty"` - VertexKeyConfig *schemas.VertexKeyConfig `gorm:"-" json:"vertex_key_config,omitempty"` - BedrockKeyConfig *schemas.BedrockKeyConfig `gorm:"-" json:"bedrock_key_config,omitempty"` - VLLMKeyConfig *schemas.VLLMKeyConfig `gorm:"-" json:"vllm_key_config,omitempty"` - ReplicateKeyConfig *schemas.ReplicateKeyConfig `gorm:"-" json:"replicate_key_config,omitempty"` - OllamaKeyConfig *schemas.OllamaKeyConfig `gorm:"-" json:"ollama_key_config,omitempty"` - SGLKeyConfig *schemas.SGLKeyConfig `gorm:"-" json:"sgl_key_config,omitempty"` + Models schemas.WhiteList `gorm:"-" json:"models"` // ["*"] allows all models; empty denies all (deny-by-default) + BlacklistedModels schemas.BlackList `gorm:"-" json:"blacklisted_models"` + Aliases schemas.KeyAliases `gorm:"-" json:"aliases,omitempty"` + AzureKeyConfig *schemas.AzureKeyConfig `gorm:"-" json:"azure_key_config,omitempty"` + VertexKeyConfig *schemas.VertexKeyConfig `gorm:"-" json:"vertex_key_config,omitempty"` + BedrockKeyConfig *schemas.BedrockKeyConfig `gorm:"-" json:"bedrock_key_config,omitempty"` + BedrockMantleKeyConfig *schemas.BedrockMantleKeyConfig `gorm:"-" json:"bedrock_mantle_key_config,omitempty"` + VLLMKeyConfig *schemas.VLLMKeyConfig `gorm:"-" json:"vllm_key_config,omitempty"` + ReplicateKeyConfig *schemas.ReplicateKeyConfig `gorm:"-" json:"replicate_key_config,omitempty"` + OllamaKeyConfig *schemas.OllamaKeyConfig `gorm:"-" json:"ollama_key_config,omitempty"` + SGLKeyConfig *schemas.SGLKeyConfig `gorm:"-" json:"sgl_key_config,omitempty"` } // TableName sets the table name for each model @@ -275,6 +285,60 @@ func (k *TableKey) BeforeSave(tx *gorm.DB) error { k.BedrockBatchS3ConfigJSON = nil } + if k.BedrockMantleKeyConfig != nil { + // Copy to avoid encrypting the shared BedrockMantleKeyConfig through the pointer. + if k.BedrockMantleKeyConfig.AccessKey.IsSet() { + ak := k.BedrockMantleKeyConfig.AccessKey + k.BedrockMantleAccessKey = &ak + } else { + k.BedrockMantleAccessKey = nil + } + if k.BedrockMantleKeyConfig.SecretKey.IsSet() { + sk := k.BedrockMantleKeyConfig.SecretKey + k.BedrockMantleSecretKey = &sk + } else { + k.BedrockMantleSecretKey = nil + } + if k.BedrockMantleKeyConfig.SessionToken != nil { + st := *k.BedrockMantleKeyConfig.SessionToken + k.BedrockMantleSessionToken = &st + } else { + k.BedrockMantleSessionToken = nil + } + if k.BedrockMantleKeyConfig.Region != nil { + br := *k.BedrockMantleKeyConfig.Region + k.BedrockMantleRegion = &br + } else { + k.BedrockMantleRegion = nil + } + if k.BedrockMantleKeyConfig.RoleARN != nil { + bra := *k.BedrockMantleKeyConfig.RoleARN + k.BedrockMantleRoleARN = &bra + } else { + k.BedrockMantleRoleARN = nil + } + if k.BedrockMantleKeyConfig.ExternalID != nil { + ei := *k.BedrockMantleKeyConfig.ExternalID + k.BedrockMantleExternalID = &ei + } else { + k.BedrockMantleExternalID = nil + } + if k.BedrockMantleKeyConfig.RoleSessionName != nil { + rsn := *k.BedrockMantleKeyConfig.RoleSessionName + k.BedrockMantleRoleSessionName = &rsn + } else { + k.BedrockMantleRoleSessionName = nil + } + } else { + k.BedrockMantleAccessKey = nil + k.BedrockMantleSecretKey = nil + k.BedrockMantleSessionToken = nil + k.BedrockMantleRegion = nil + k.BedrockMantleRoleARN = nil + k.BedrockMantleExternalID = nil + k.BedrockMantleRoleSessionName = nil + } + if k.Aliases != nil { data, err := sonic.Marshal(k.Aliases) if err != nil { @@ -396,6 +460,28 @@ func (k *TableKey) BeforeSave(tx *gorm.DB) error { if err := encryptString(k.BedrockBatchS3ConfigJSON); err != nil { return fmt.Errorf("failed to encrypt bedrock batch s3 config: %w", err) } + // Bedrock Mantle + if err := encryptSecretVarPtr(&k.BedrockMantleAccessKey); err != nil { + return fmt.Errorf("failed to encrypt bedrock mantle access key: %w", err) + } + if err := encryptSecretVarPtr(&k.BedrockMantleSecretKey); err != nil { + return fmt.Errorf("failed to encrypt bedrock mantle secret key: %w", err) + } + if err := encryptSecretVarPtr(&k.BedrockMantleSessionToken); err != nil { + return fmt.Errorf("failed to encrypt bedrock mantle session token: %w", err) + } + if err := encryptSecretVarPtr(&k.BedrockMantleRegion); err != nil { + return fmt.Errorf("failed to encrypt bedrock mantle region: %w", err) + } + if err := encryptSecretVarPtr(&k.BedrockMantleRoleARN); err != nil { + return fmt.Errorf("failed to encrypt bedrock mantle role arn: %w", err) + } + if err := encryptSecretVarPtr(&k.BedrockMantleExternalID); err != nil { + return fmt.Errorf("failed to encrypt bedrock mantle external id: %w", err) + } + if err := encryptSecretVarPtr(&k.BedrockMantleRoleSessionName); err != nil { + return fmt.Errorf("failed to encrypt bedrock mantle role session name: %w", err) + } // Aliases if err := encryptString(k.AliasesJSON); err != nil { return fmt.Errorf("failed to encrypt aliases: %w", err) @@ -480,6 +566,28 @@ func (k *TableKey) AfterFind(tx *gorm.DB) error { if err := decryptString(k.BedrockBatchS3ConfigJSON); err != nil { return fmt.Errorf("failed to decrypt bedrock batch s3 config: %w", err) } + // Bedrock Mantle + if err := decryptSecretVarPtr(&k.BedrockMantleAccessKey); err != nil { + return fmt.Errorf("failed to decrypt bedrock mantle access key: %w", err) + } + if err := decryptSecretVarPtr(&k.BedrockMantleSecretKey); err != nil { + return fmt.Errorf("failed to decrypt bedrock mantle secret key: %w", err) + } + if err := decryptSecretVarPtr(&k.BedrockMantleSessionToken); err != nil { + return fmt.Errorf("failed to decrypt bedrock mantle session token: %w", err) + } + if err := decryptSecretVarPtr(&k.BedrockMantleRegion); err != nil { + return fmt.Errorf("failed to decrypt bedrock mantle region: %w", err) + } + if err := decryptSecretVarPtr(&k.BedrockMantleRoleARN); err != nil { + return fmt.Errorf("failed to decrypt bedrock mantle role arn: %w", err) + } + if err := decryptSecretVarPtr(&k.BedrockMantleExternalID); err != nil { + return fmt.Errorf("failed to decrypt bedrock mantle external id: %w", err) + } + if err := decryptSecretVarPtr(&k.BedrockMantleRoleSessionName); err != nil { + return fmt.Errorf("failed to decrypt bedrock mantle role session name: %w", err) + } // Aliases if err := decryptString(k.AliasesJSON); err != nil { return fmt.Errorf("failed to decrypt aliases: %w", err) @@ -587,6 +695,22 @@ func (k *TableKey) AfterFind(tx *gorm.DB) error { k.BedrockKeyConfig = bedrockConfig } + // Reconstruct Bedrock Mantle config if fields are present + if k.BedrockMantleAccessKey != nil || k.BedrockMantleSecretKey != nil || k.BedrockMantleSessionToken != nil || k.BedrockMantleRegion != nil || k.BedrockMantleRoleARN != nil || k.BedrockMantleExternalID != nil || k.BedrockMantleRoleSessionName != nil { + mantleConfig := &schemas.BedrockMantleKeyConfig{} + if k.BedrockMantleAccessKey != nil { + mantleConfig.AccessKey = *k.BedrockMantleAccessKey + } + if k.BedrockMantleSecretKey != nil { + mantleConfig.SecretKey = *k.BedrockMantleSecretKey + } + mantleConfig.SessionToken = k.BedrockMantleSessionToken + mantleConfig.Region = k.BedrockMantleRegion + mantleConfig.RoleARN = k.BedrockMantleRoleARN + mantleConfig.ExternalID = k.BedrockMantleExternalID + mantleConfig.RoleSessionName = k.BedrockMantleRoleSessionName + k.BedrockMantleKeyConfig = mantleConfig + } // Reconstruct Aliases if k.AliasesJSON != nil && *k.AliasesJSON != "" { var aliases schemas.KeyAliases diff --git a/framework/configstore/tables/mcp.go b/framework/configstore/tables/mcp.go index 38d1a59f996..cd6c2f79e07 100644 --- a/framework/configstore/tables/mcp.go +++ b/framework/configstore/tables/mcp.go @@ -28,6 +28,7 @@ type TableMCPClient struct { IsPingAvailable *bool `gorm:"default:true" json:"is_ping_available,omitempty"` // Whether the MCP server supports ping for health checks ToolPricingJSON string `gorm:"type:text" json:"-"` // JSON serialized map[string]float64 ToolSyncInterval int `gorm:"default:0" json:"tool_sync_interval"` // Per-client tool sync interval in seconds (0 = use global, negative = disabled) + ToolExecutionTimeout int `gorm:"default:0" json:"tool_execution_timeout"` // Per-client tool execution timeout in seconds (0 = use global from tool_manager_config) // Per-user OAuth: discovered tools persisted so they survive restart DiscoveredToolsJSON string `gorm:"type:text" json:"-"` // JSON serialized map[string]schemas.ChatTool diff --git a/framework/configstore/tables/mcp_per_user_headers.go b/framework/configstore/tables/mcpheaders.go similarity index 100% rename from framework/configstore/tables/mcp_per_user_headers.go rename to framework/configstore/tables/mcpheaders.go diff --git a/framework/configstore/tables/mcp_library.go b/framework/configstore/tables/mcplibrary.go similarity index 100% rename from framework/configstore/tables/mcp_library.go rename to framework/configstore/tables/mcplibrary.go diff --git a/framework/configstore/tables/oauth.go b/framework/configstore/tables/mcpoauth2.go similarity index 84% rename from framework/configstore/tables/oauth.go rename to framework/configstore/tables/mcpoauth2.go index 3012baeeed6..652d0d902d1 100644 --- a/framework/configstore/tables/oauth.go +++ b/framework/configstore/tables/mcpoauth2.go @@ -12,26 +12,26 @@ import ( // TableOauthConfig represents an OAuth configuration in the database // This stores the OAuth client configuration and flow state type TableOauthConfig struct { - ID string `gorm:"type:varchar(255);primaryKey" json:"id"` // UUID + ID string `gorm:"type:varchar(255);primaryKey" json:"id"` // UUID ClientID *schemas.SecretVar `gorm:"type:varchar(512)" json:"client_id"` // OAuth provider's client ID (optional for public clients) ClientSecret *schemas.SecretVar `gorm:"type:text" json:"-"` // Encrypted OAuth client secret (optional for public clients) - AuthorizeURL string `gorm:"type:text" json:"authorize_url"` // Provider's authorization endpoint (optional, can be discovered) - TokenURL string `gorm:"type:text" json:"token_url"` // Provider's token endpoint (optional, can be discovered) - RegistrationURL *string `gorm:"type:text" json:"registration_url,omitempty"` // Provider's dynamic registration endpoint (optional, can be discovered) - RedirectURI string `gorm:"type:text;not null" json:"redirect_uri"` // Callback URL - Scopes string `gorm:"type:text" json:"scopes"` // JSON array of scopes (optional, can be discovered) - State string `gorm:"type:varchar(255);uniqueIndex;not null" json:"-"` // CSRF state token - CodeVerifier string `gorm:"type:text" json:"-"` // PKCE code verifier (generated, kept secret) - CodeChallenge string `gorm:"type:varchar(255)" json:"code_challenge"` // PKCE code challenge (sent to provider) - Status string `gorm:"type:varchar(50);not null;index" json:"status"` // "pending", "authorized", "failed", "expired", "revoked" - TokenID *string `gorm:"type:varchar(255);index" json:"token_id"` // Foreign key to oauth_tokens.ID (set after callback) - ServerURL string `gorm:"type:text" json:"server_url"` // MCP server URL for OAuth discovery - UseDiscovery bool `gorm:"default:false" json:"use_discovery"` // Flag to enable OAuth discovery - MCPClientConfigJSON *string `gorm:"type:text" json:"-"` // JSON serialized MCPClientConfig for multi-instance support (pending MCP client waiting for OAuth completion) - EncryptionStatus string `gorm:"type:varchar(20);default:'plain_text'" json:"-"` - CreatedAt time.Time `gorm:"index;not null" json:"created_at"` - UpdatedAt time.Time `gorm:"index;not null" json:"updated_at"` - ExpiresAt time.Time `gorm:"index;not null" json:"expires_at"` // State expiry (15 min) + AuthorizeURL string `gorm:"type:text" json:"authorize_url"` // Provider's authorization endpoint (optional, can be discovered) + TokenURL string `gorm:"type:text" json:"token_url"` // Provider's token endpoint (optional, can be discovered) + RegistrationURL *string `gorm:"type:text" json:"registration_url,omitempty"` // Provider's dynamic registration endpoint (optional, can be discovered) + RedirectURI string `gorm:"type:text;not null" json:"redirect_uri"` // Callback URL + Scopes string `gorm:"type:text" json:"scopes"` // JSON array of scopes (optional, can be discovered) + State string `gorm:"type:varchar(255);uniqueIndex;not null" json:"-"` // CSRF state token + CodeVerifier string `gorm:"type:text" json:"-"` // PKCE code verifier (generated, kept secret) + CodeChallenge string `gorm:"type:varchar(255)" json:"code_challenge"` // PKCE code challenge (sent to provider) + Status string `gorm:"type:varchar(50);not null;index" json:"status"` // "pending", "authorized", "failed", "expired", "revoked" + TokenID *string `gorm:"type:varchar(255);index" json:"token_id"` // Foreign key to oauth_tokens.ID (set after callback) + ServerURL string `gorm:"type:text" json:"server_url"` // MCP server URL for OAuth discovery + UseDiscovery bool `gorm:"default:false" json:"use_discovery"` // Flag to enable OAuth discovery + MCPClientConfigJSON *string `gorm:"type:text" json:"-"` // JSON serialized MCPClientConfig for multi-instance support (pending MCP client waiting for OAuth completion) + EncryptionStatus string `gorm:"type:varchar(20);default:'plain_text'" json:"-"` + CreatedAt time.Time `gorm:"index;not null" json:"created_at"` + UpdatedAt time.Time `gorm:"index;not null" json:"updated_at"` + ExpiresAt time.Time `gorm:"index;not null" json:"expires_at"` // State expiry (15 min) } // TableName sets the table name diff --git a/framework/configstore/tables/mcpoauth2issuance.go b/framework/configstore/tables/mcpoauth2issuance.go new file mode 100644 index 00000000000..bc34a6e62dd --- /dev/null +++ b/framework/configstore/tables/mcpoauth2issuance.go @@ -0,0 +1,129 @@ +package tables + +import ( + "encoding/json" + "time" + + "gorm.io/gorm" +) + +// TableOAuth2Client holds a registered OAuth2 client created via Dynamic Client +// Registration (RFC 7591). Bifrost only supports public clients +// (token_endpoint_auth_method=none) — no client secrets. +type TableOAuth2Client struct { + ID string `gorm:"type:varchar(255);primaryKey" json:"id"` + ClientID string `gorm:"type:varchar(255);uniqueIndex;not null" json:"client_id"` + ClientName string `gorm:"type:varchar(255)" json:"client_name"` + RedirectURIsJSON string `gorm:"type:text;not null" json:"-"` // JSON []string + GrantTypesJSON string `gorm:"type:text;not null" json:"-"` // JSON []string + Scope string `gorm:"type:varchar(255)" json:"scope"` + CreatedAt time.Time `gorm:"index;not null" json:"created_at"` + + // Virtual fields + RedirectURIs []string `gorm:"-" json:"redirect_uris"` + GrantTypes []string `gorm:"-" json:"grant_types"` +} + +func (TableOAuth2Client) TableName() string { return "oauth2_clients" } + +func (c *TableOAuth2Client) BeforeSave(tx *gorm.DB) error { + if c.RedirectURIs != nil { + data, err := json.Marshal(c.RedirectURIs) + if err != nil { + return err + } + c.RedirectURIsJSON = string(data) + } + if c.GrantTypes != nil { + data, err := json.Marshal(c.GrantTypes) + if err != nil { + return err + } + c.GrantTypesJSON = string(data) + } + return nil +} + +func (c *TableOAuth2Client) AfterFind(tx *gorm.DB) error { + if c.RedirectURIsJSON != "" { + if err := json.Unmarshal([]byte(c.RedirectURIsJSON), &c.RedirectURIs); err != nil { + return err + } + } + if c.GrantTypesJSON != "" { + if err := json.Unmarshal([]byte(c.GrantTypesJSON), &c.GrantTypes); err != nil { + return err + } + } + return nil +} + +// OAuth2AuthorizeRequestStatus is the status of a downstream authorize request. +type OAuth2AuthorizeRequestStatus string + +const ( + OAuth2AuthorizeRequestStatusPending OAuth2AuthorizeRequestStatus = "pending" // waiting for consent + OAuth2AuthorizeRequestStatusConsented OAuth2AuthorizeRequestStatus = "consented" // identity resolved, code minted + OAuth2AuthorizeRequestStatusCodeIssued OAuth2AuthorizeRequestStatus = "code_issued" // token exchanged, one-time consumed +) + +// TableOAuth2AuthorizeRequest tracks a pending downstream OAuth2 authorization +// request from creation at /oauth2/authorize through consent to token exchange +// at /oauth2/token. +// +// State transitions: +// - pending — request created; browser redirected to consent page +// - consented — user approved; identity resolved; auth code minted (CodeHash set) +// - code_issued — auth code exchanged at /oauth2/token; row is consumed (single-use) +type TableOAuth2AuthorizeRequest struct { + ID string `gorm:"type:varchar(255);primaryKey" json:"id"` + ClientID string `gorm:"type:varchar(255);not null;index" json:"client_id"` + RedirectURI string `gorm:"type:text;not null" json:"-"` + State string `gorm:"type:varchar(512);not null" json:"-"` // CSRF; returned in redirect + Scope string `gorm:"type:varchar(255)" json:"scope"` + Resource string `gorm:"type:text;not null" json:"-"` // RFC 8707 resource indicator + CodeChallenge string `gorm:"type:varchar(512);not null" json:"-"` // PKCE S256 challenge + CodeChallengeMethod string `gorm:"type:varchar(10);not null" json:"-"` // always "S256" + Status OAuth2AuthorizeRequestStatus `gorm:"type:varchar(20);not null;index" json:"status"` + // Set by the consent flow once the user approves: + BfMode string `gorm:"type:varchar(20)" json:"bf_mode,omitempty"` // user|vk|session + BfSub string `gorm:"type:varchar(255)" json:"bf_sub,omitempty"` // resolved identity + // nil while pending; set to SHA256(auth_code) at consent. A pointer so unset + // rows store SQL NULL — NULLs are distinct under the unique index, letting many + // requests stay pending at once while still enforcing uniqueness for real hashes. + CodeHash *string `gorm:"type:varchar(255);uniqueIndex" json:"-"` + // TTL: + ExpiresAt time.Time `gorm:"index;not null" json:"expires_at"` + CreatedAt time.Time `gorm:"not null" json:"created_at"` + UpdatedAt time.Time `gorm:"not null" json:"updated_at"` +} + +func (TableOAuth2AuthorizeRequest) TableName() string { return "oauth2_authorize_requests" } + +// TableOAuth2RefreshToken stores a hashed rotating refresh token. The plaintext +// token is only returned to the client once at issuance; only the SHA256 hash +// is persisted. Invalidation paths: +// - rotation on use: old token revoked atomically when a new one is issued +// - bf_sub liveness: VK deleted or user deactivated → invalid_grant on next refresh +// - explicit revocation via the Connected Clients UI +// +// FamilyID links all tokens descended from the same authorization grant (set to +// the authorize request ID at first issuance, propagated on every rotation). +// When a revoked token is presented — indicating the token was stolen and used +// after the legitimate client already rotated — all tokens sharing the FamilyID +// are revoked immediately, per the OAuth 2.0 Security BCP (RFC 9700 §2.2.2). +type TableOAuth2RefreshToken struct { + ID string `gorm:"type:varchar(255);primaryKey" json:"id"` + TokenHash string `gorm:"type:varchar(255);uniqueIndex;not null" json:"-"` // SHA256 hex + FamilyID string `gorm:"type:varchar(255);not null;index" json:"family_id"` // authorize request ID + ClientID string `gorm:"type:varchar(255);not null;index" json:"client_id"` + BfMode string `gorm:"type:varchar(20);not null" json:"bf_mode"` // user|vk|session + BfSub string `gorm:"type:varchar(255);not null" json:"bf_sub"` // resolved identity + Scope string `gorm:"type:varchar(255)" json:"scope"` + Resource string `gorm:"type:text;not null" json:"-"` // RFC 8707 resource indicator; preserved across rotations for the JWT aud claim + RevokedAt *time.Time `gorm:"index" json:"revoked_at,omitempty"` + LastUsedAt *time.Time `gorm:"index" json:"last_used_at,omitempty"` + CreatedAt time.Time `gorm:"not null" json:"created_at"` +} + +func (TableOAuth2RefreshToken) TableName() string { return "oauth2_refresh_tokens" } diff --git a/framework/configstore/tables/mcpoauth2server.go b/framework/configstore/tables/mcpoauth2server.go new file mode 100644 index 00000000000..b725ebd7d34 --- /dev/null +++ b/framework/configstore/tables/mcpoauth2server.go @@ -0,0 +1,140 @@ +package tables + +import ( + "fmt" + + "github.com/maximhq/bifrost/core/schemas" + "github.com/maximhq/bifrost/framework/encrypt" +) + +// MCPServerAuthMode controls how Bifrost's /mcp endpoint authenticates inbound +// MCP clients. It does not affect how Bifrost authenticates to upstream MCP +// servers (governed by MCPClientConfig.AuthType). +type MCPServerAuthMode string + +const ( + DefaultAuthCodeTTL = 300 // 5 minutes + MaxAuthCodeTTL = 900 // 15 minutes — hard ceiling; enforced at config save and at code issuance + DefaultAccessTokenTTL = 600 // 10 minutes +) + +const ( + // MCPServerAuthModeHeaders accepts header credentials only: x-bf-vk, + // Authorization: Bearer , x-api-key, x-bf-mcp-session-id. + // Discovery endpoints return 404. Default — today's behavior. + MCPServerAuthModeHeaders MCPServerAuthMode = "headers" + + // MCPServerAuthModeBoth accepts both header credentials and Bifrost-issued + // JWTs. Discovery endpoints are live; existing header-credential clients + // that never receive a 401 are unaffected. + MCPServerAuthModeBoth MCPServerAuthMode = "both" + + // MCPServerAuthModeOAuth accepts Bifrost-issued JWTs only. Header + // credentials (VK / api-key / session) are rejected on /mcp. + // WARNING: existing virtual-key MCP integrations will stop working. + MCPServerAuthModeOAuth MCPServerAuthMode = "oauth" +) + +// OAuth2ServerConfig holds the OAuth2 authorization-server settings serialized +// as JSON into config_client.oauth2_server_config_json. Only meaningful when +// MCPServerAuthMode is MCPServerAuthModeBoth or MCPServerAuthModeOAuth. +// Not a table of its own. +type OAuth2ServerConfig struct { + // IssuerURL is Bifrost's OAuth authorization-server identity — it appears + // as the `issuer` in discovery documents and as the `iss` claim in every + // issued JWT. Supports env var syntax ("env.MY_VAR"). When empty, + // BuildBaseURL(request) is used as a per-request fallback, which works for + // single-host / dev deployments. Multi-host or reverse-proxy deployments + // MUST set a stable value; token verification fails when the Host header + // differs across nodes. + IssuerURL *schemas.SecretVar `json:"issuer_url,omitempty"` + + // AuthCodeTTL is the lifetime of the one-time authorization code issued by + // /oauth2/authorize and exchanged at /oauth2/token (seconds, default 300). + // The code is single-use — it is invalidated the moment it is exchanged or + // expires. Short TTL is intentional: if the window lapses the user simply + // re-authenticates. Capped at MaxAuthCodeTTL (900s); larger stored values + // are rejected at config save and clamped to the cap at issuance. + AuthCodeTTL int `json:"auth_code_ttl"` + + // AccessTokenTTL is the lifetime of the issued JWT Bearer token (seconds, + // default 600 = 10 min). When the token expires the client uses its refresh + // token to silently obtain a new one without any user interaction. + AccessTokenTTL int `json:"access_token_ttl"` + + // DisableVKIdentity removes virtual-key identity from the OAuth consent flow: + // vk is neither offered on the consent page nor accepted if submitted, and + // existing vk-mode grants are rejected immediately at request time and cannot + // refresh. Anonymous session identity is unaffected — that is governed + // separately by EnforceAuthOnInference. Honored only when an identity provider + // is configured, so it can never strip the consent flow of every identity + // option. Only meaningful when MCPServerAuthMode is oauth. + DisableVKIdentity bool `json:"disable_vk_identity,omitempty"` + + // Refresh tokens have no hard expiry — they are invalidated only by: + // - rotation on use (each /oauth2/token refresh call issues a new token + // and immediately invalidates the previous one) + // - bf_sub liveness check on refresh (VK or user deleted / deactivated → + // invalid_grant, forcing re-authentication) + // - explicit revocation via the OAuth Grants UI + // - DisableVKIdentity enabled (vk-mode grants denied on refresh) + // No RefreshTokenTTL field exists by design — there is no timer, only + // explicit invalidation paths. +} + +// DefaultOAuth2ServerConfig returns sensible defaults for the AS-specific settings. +func DefaultOAuth2ServerConfig() *OAuth2ServerConfig { + return &OAuth2ServerConfig{ + AuthCodeTTL: DefaultAuthCodeTTL, + AccessTokenTTL: DefaultAccessTokenTTL, + } +} + +// OAuth2SigningKey holds the single RS256 keypair used to sign Bifrost-issued +// JWTs. Stored as JSON in governance_config under GovernanceConfigKeyOAuth2SigningKey. +// The private key PEM is encrypted via framework/encrypt before storage. +type OAuth2SigningKey struct { + KID string `json:"kid"` // key ID embedded in JWT headers + PrivateKeyPEM string `json:"private_key_pem"` // encrypted at rest via framework/encrypt when EncryptionStatus is "encrypted" + PublicKeyPEM string `json:"public_key_pem"` // plaintext; public key is not sensitive + EncryptionStatus string `json:"encryption_status,omitempty"` // EncryptionStatusPlainText or EncryptionStatusEncrypted — records whether PrivateKeyPEM was encrypted at write time so reads do not depend on the current encrypt.IsEnabled() state +} + +// Encrypt encrypts PrivateKeyPEM in place and stamps EncryptionStatus, mirroring +// the BeforeSave hooks on secret-bearing tables. Because the signing key is +// persisted as a JSON blob inside governance_config (not its own GORM row) it +// cannot rely on GORM hooks, so callers invoke this before marshaling/storing. +func (k *OAuth2SigningKey) Encrypt() error { + if encrypt.IsEnabled() && k.PrivateKeyPEM != "" { + if err := encryptString(&k.PrivateKeyPEM); err != nil { + return fmt.Errorf("failed to encrypt oauth2 signing key: %w", err) + } + k.EncryptionStatus = EncryptionStatusEncrypted + } else { + k.EncryptionStatus = EncryptionStatusPlainText + } + return nil +} + +// Decrypt decrypts PrivateKeyPEM in place based on the stored EncryptionStatus +// marker, mirroring the AfterFind hooks on secret-bearing tables. Keys written +// before encryption was enabled are marked plain_text and returned as-is. +// Records written before this marker existed carry an empty status and fall back +// to the historical encrypt.IsEnabled() behavior. +func (k *OAuth2SigningKey) Decrypt() error { + if k.PrivateKeyPEM == "" { + return nil + } + shouldDecrypt := k.EncryptionStatus == EncryptionStatusEncrypted || + (k.EncryptionStatus == "" && encrypt.IsEnabled()) + if shouldDecrypt { + if err := decryptString(&k.PrivateKeyPEM); err != nil { + return fmt.Errorf("failed to decrypt oauth2 signing key: %w", err) + } + } + return nil +} + +// GovernanceConfigKeyOAuth2SigningKey is the governance_config key under which +// the OAuth2 signing keypair is stored. +const GovernanceConfigKeyOAuth2SigningKey = "oauth2_signing_key" diff --git a/framework/configstore/tables/modelpricing.go b/framework/configstore/tables/modelpricing.go index 2b015daf901..d1c7385dd7b 100644 --- a/framework/configstore/tables/modelpricing.go +++ b/framework/configstore/tables/modelpricing.go @@ -18,6 +18,7 @@ type TableModelPricing struct { MaxInputTokens *int `gorm:"default:null" json:"max_input_tokens,omitempty"` MaxOutputTokens *int `gorm:"default:null" json:"max_output_tokens,omitempty"` Architecture *schemas.Architecture `gorm:"type:text;serializer:json;default:null" json:"architecture,omitempty"` + IsDeprecated bool `gorm:"default:false;column:is_deprecated" json:"is_deprecated"` // Costs - Text InputCostPerToken *float64 `gorm:"default:null" json:"input_cost_per_token,omitempty"` diff --git a/framework/configstore/tables/routing_rules.go b/framework/configstore/tables/routingrules.go similarity index 100% rename from framework/configstore/tables/routing_rules.go rename to framework/configstore/tables/routingrules.go diff --git a/framework/configstore/tables/temp_token.go b/framework/configstore/tables/temptokens.go similarity index 100% rename from framework/configstore/tables/temp_token.go rename to framework/configstore/tables/temptokens.go diff --git a/framework/configstore/tables/virtualkey.go b/framework/configstore/tables/virtualkey.go index c12371f6b5c..8fc007d0991 100644 --- a/framework/configstore/tables/virtualkey.go +++ b/framework/configstore/tables/virtualkey.go @@ -150,6 +150,16 @@ func (pc *TableVirtualKeyProviderConfig) AfterFind(tx *gorm.DB) error { key.BedrockRoleSessionName = nil key.BedrockKeyConfig = nil + // Clear all Bedrock Mantle-related sensitive fields + key.BedrockMantleAccessKey = nil + key.BedrockMantleSecretKey = nil + key.BedrockMantleSessionToken = nil + key.BedrockMantleRegion = nil + key.BedrockMantleRoleARN = nil + key.BedrockMantleExternalID = nil + key.BedrockMantleRoleSessionName = nil + key.BedrockMantleKeyConfig = nil + pc.Keys[i] = *key } } @@ -209,8 +219,9 @@ type TableVirtualKey struct { ID string `gorm:"primaryKey;type:varchar(255)" json:"id"` Name string `gorm:"uniqueIndex:idx_virtual_key_name;type:varchar(255);not null" json:"name"` Description string `gorm:"type:text" json:"description,omitempty"` - Value string `gorm:"uniqueIndex:idx_virtual_key_value;type:text;not null" json:"value"` + Value schemas.SecretVar `gorm:"uniqueIndex:idx_virtual_key_value;type:text;not null" json:"value"` IsActive *bool `gorm:"default:true" json:"is_active,omitempty"` // Nil means true (DB default); false means inactive + ExpiresAt *time.Time `gorm:"type:timestamp;null" json:"expires_at,omitempty"` // Optional expiry; nil means never expires ProviderConfigs []TableVirtualKeyProviderConfig `gorm:"foreignKey:VirtualKeyID;constraint:OnDelete:CASCADE" json:"provider_configs"` // Empty means no providers allowed (deny-by-default) MCPConfigs []TableVirtualKeyMCPConfig `gorm:"foreignKey:VirtualKeyID;constraint:OnDelete:CASCADE" json:"mcp_configs"` @@ -254,6 +265,37 @@ func (vk *TableVirtualKey) IsActiveValue() bool { return *vk.IsActive } +// VaultPathKey implements schemas.VaultPathKeyer so vault callbacks can compute the +// vault base path for this model automatically. +func (vk *TableVirtualKey) VaultPathKey() string { return vk.ID } + +// VaultStoreSelfManaged marks TableVirtualKey as storing its own vault secrets from +// within BeforeSave, so the global vault callback skips it. +func (vk *TableVirtualKey) VaultStoreSelfManaged() {} + +// MarshalJSON serializes TableVirtualKey with Value emitted as a resolved plain string, +// never as a SecretVar object. This ensures all REST API responses return "bfvk-xxx" +// rather than {"value":"bfvk-xxx","type":"plain_text"}. +func (vk TableVirtualKey) MarshalJSON() ([]byte, error) { + type Alias TableVirtualKey + return json.Marshal(&struct { + Alias + Value string `json:"value"` + }{ + Alias: Alias(vk), + Value: vk.Value.GetValue(), + }) +} + +// IsExpiredAt reports whether the virtual key has passed its expiry. +// now == expires_at is treated as expired; nil ExpiresAt means never expires. +func (vk *TableVirtualKey) IsExpiredAt(now time.Time) bool { + if vk == nil || vk.ExpiresAt == nil { + return false + } + return !now.UTC().Before(vk.ExpiresAt.UTC()) +} + // BeforeSave is a GORM hook that enforces mutual exclusion (team vs customer), computes // a SHA-256 hash of the plaintext value for indexed lookups, and encrypts the virtual key // value before writing to the database. @@ -263,13 +305,23 @@ func (vk *TableVirtualKey) BeforeSave(tx *gorm.DB) error { return fmt.Errorf("virtual key cannot belong to both team and customer") } - // Hash must be computed before encryption (from plaintext value) - if vk.Value != "" { - vk.ValueHash = encrypt.HashSHA256(vk.Value) - + // Hash must be computed before encryption (from plaintext value). + if vk.Value.IsSet() { + resolved := vk.Value.GetValue() + if resolved == "" { + return fmt.Errorf("virtual key %s: env/vault ref %q could not be resolved", vk.ID, vk.Value.GetRawRef()) + } + vk.ValueHash = encrypt.HashSHA256(resolved) + } + // Store plaintext SecretVar into vault and rewrite to vault ref before encrypting. + if schemas.VaultStoreWriteEnabled() { + base := schemas.VaultBasePath(vk.TableName(), vk.VaultPathKey()) + if err := schemas.StoreOwnedVaultSecretVars(tx.Statement.Context, base, vk); err != nil { + return fmt.Errorf("failed to store virtual key secrets to vault: %w", err) + } } - if encrypt.IsEnabled() && vk.Value != "" { - if err := encryptString(&vk.Value); err != nil { + if encrypt.IsEnabled() && vk.Value.IsSet() { + if err := encryptSecretVar(&vk.Value); err != nil { return fmt.Errorf("failed to encrypt virtual key value: %w", err) } vk.EncryptionStatus = EncryptionStatusEncrypted @@ -285,7 +337,7 @@ func (vk *TableVirtualKey) BeforeSave(tx *gorm.DB) error { func (vk *TableVirtualKey) AfterFind(tx *gorm.DB) error { switch vk.EncryptionStatus { case EncryptionStatusEncrypted: - if err := decryptString(&vk.Value); err != nil { + if err := decryptSecretVar(&vk.Value); err != nil { return fmt.Errorf("failed to decrypt virtual key value: %w", err) } } diff --git a/framework/configstore/vault_callbacks.go b/framework/configstore/vault_callbacks.go index 3f5f0df7768..b2582bef9a6 100644 --- a/framework/configstore/vault_callbacks.go +++ b/framework/configstore/vault_callbacks.go @@ -1,6 +1,7 @@ package configstore import ( + "log" "reflect" "github.com/maximhq/bifrost/core/schemas" @@ -59,7 +60,10 @@ func vaultRemoveCallback(tx *gorm.DB) { forEachModel(tx, func(model interface{}, keyer schemas.VaultPathKeyer) { tableName := tx.Statement.Table base := schemas.VaultBasePath(tableName, keyer.VaultPathKey()) - schemas.RemoveOwnedVaultSecretVars(tx.Statement.Context, base, model) + errs := schemas.RemoveOwnedVaultSecretVars(tx.Statement.Context, base, model) + for _, err := range errs { + log.Printf("vault: failed to remove secret for %s/%s: %v", tableName, keyer.VaultPathKey(), err) + } }) } diff --git a/framework/configstore/vault_callbacks_test.go b/framework/configstore/vault_callbacks_test.go index 1da5fa52e83..7cb3b176597 100644 --- a/framework/configstore/vault_callbacks_test.go +++ b/framework/configstore/vault_callbacks_test.go @@ -136,6 +136,45 @@ func TestVaultCallbacks_SelfManagedStoresPlaintext(t *testing.T) { } } +// TestVaultCallbacks_SelfManagedRemovesVaultSecrets verifies that deleting a TableKey +// (a self-managed model) removes all its owned vault secrets via the global remove callback. +func TestVaultCallbacks_SelfManagedRemovesVaultSecrets(t *testing.T) { + stored, removed := stubVaultHooks(t) + + db, err := gorm.Open(sqlite.Open(":memory:"), &gorm.Config{}) + require.NoError(t, err) + RegisterVaultCallbacks(db) + require.NoError(t, db.AutoMigrate(&tables.TableKey{})) + + keyID := "key-delete-test" + key := &tables.TableKey{ + Name: "k-delete", + KeyID: keyID, + Provider: "openai", + Value: schemas.SecretVar{Val: "sk-secret-value"}, + Models: schemas.WhiteList{"*"}, + } + require.NoError(t, db.Create(key).Error) + + valuePath := fmt.Sprintf("bifrost/config_keys/%s/value", keyID) + require.Equal(t, "sk-secret-value", stored[valuePath], "value not stored to vault on create") + + // Load fresh to get the vault ref populated in SecretVar fields. + var toDelete tables.TableKey + require.NoError(t, db.First(&toDelete, "key_id = ?", keyID).Error) + require.Equal(t, "vault."+valuePath, toDelete.Value.GetRawRef(), "loaded key should carry vault ref") + + require.NoError(t, db.Delete(&toDelete).Error) + + found := false + for _, p := range *removed { + if p == valuePath { + found = true + } + } + require.True(t, found, "expected vault remove for %q, got %v", valuePath, *removed) +} + func TestVaultCallbacks_NoOpWhenDisabled(t *testing.T) { // No hooks installed -> VaultStoreEnabled() is false -> callbacks no-op. prevStore := schemas.VaultStoreHook diff --git a/framework/docker-compose.yml b/framework/docker-compose.yml index 9845e1b70fb..4585fbb97a1 100644 --- a/framework/docker-compose.yml +++ b/framework/docker-compose.yml @@ -8,6 +8,11 @@ # For production, use cloud service with PINECONE_API_KEY and PINECONE_INDEX_HOST # See: https://docs.pinecone.io/guides/operations/local-development # +# Supported Log Stores: +# - Postgres: shared with configstore (port 5432) +# - ClickHouse: native protocol on host port 9001 (container 9000; host 9000 is +# taken by Weaviate), HTTP on 8123. Used by logstore clickhouse tests. +# services: postgres: image: postgres:16-alpine @@ -30,6 +35,31 @@ services: networks: - bifrost_network + clickhouse: + image: clickhouse/clickhouse-server:24.8-alpine + container_name: bifrost-clickhouse-fw + environment: + CLICKHOUSE_DB: bifrost + CLICKHOUSE_USER: bifrost + CLICKHOUSE_PASSWORD: bifrost_password + ports: + - "9001:9000" + - "8123:8123" + volumes: + - clickhouse_data:/var/lib/clickhouse + ulimits: + nofile: + soft: 262144 + hard: 262144 + healthcheck: + test: ["CMD", "wget", "--spider", "-q", "http://127.0.0.1:8123/ping"] + interval: 10s + timeout: 5s + retries: 5 + restart: unless-stopped + networks: + - bifrost_network + redis: image: redis/redis-stack:latest container_name: bifrost-redis @@ -112,6 +142,8 @@ networks: volumes: postgres_data: driver: local + clickhouse_data: + driver: local weaviate_data: driver: local redis_data: diff --git a/framework/go.mod b/framework/go.mod index e1793679965..da05679f34c 100644 --- a/framework/go.mod +++ b/framework/go.mod @@ -18,6 +18,7 @@ require ( golang.org/x/crypto v0.52.0 golang.org/x/sync v0.20.0 google.golang.org/api v0.282.0 + gorm.io/driver/clickhouse v0.7.0 gorm.io/driver/postgres v1.6.0 gorm.io/driver/sqlite v1.6.0 gorm.io/gorm v1.31.1 @@ -34,6 +35,8 @@ require ( github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.13.1 // indirect github.com/Azure/azure-sdk-for-go/sdk/internal v1.11.2 // indirect github.com/AzureAD/microsoft-authentication-library-for-go v1.6.0 // indirect + github.com/ClickHouse/ch-go v0.61.5 // indirect + github.com/ClickHouse/clickhouse-go/v2 v2.30.0 // indirect github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.31.0 // indirect github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.55.0 // indirect github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/resourcemapping v0.55.0 // indirect @@ -49,6 +52,8 @@ require ( github.com/envoyproxy/go-control-plane/envoy v1.37.0 // indirect github.com/envoyproxy/protoc-gen-validate v1.3.3 // indirect github.com/felixge/httpsnoop v1.0.4 // indirect + github.com/go-faster/city v1.0.1 // indirect + github.com/go-faster/errors v0.7.1 // indirect github.com/go-jose/go-jose/v4 v4.1.4 // indirect github.com/go-logr/logr v1.4.3 // indirect github.com/go-logr/stdr v1.2.2 // indirect @@ -69,13 +74,18 @@ require ( github.com/google/s2a-go v0.1.9 // indirect github.com/googleapis/enterprise-certificate-proxy v0.3.16 // indirect github.com/googleapis/gax-go/v2 v2.22.0 // indirect + github.com/hashicorp/go-version v1.6.0 // indirect github.com/jackc/pgpassfile v1.0.0 // indirect github.com/jackc/pgservicefile v0.0.0-20240606120523-5a60cdf6a761 // indirect github.com/jackc/puddle/v2 v2.2.2 // indirect github.com/kylelemons/godebug v1.1.0 // indirect github.com/oapi-codegen/runtime v1.1.1 // indirect + github.com/paulmach/orb v0.11.1 // indirect + github.com/pierrec/lz4/v4 v4.1.21 // indirect github.com/pkg/browser v0.0.0-20240102092130-5ac0b6a4141c // indirect github.com/planetscale/vtprotobuf v0.6.1-0.20240319094008-0393e58bdf10 // indirect + github.com/segmentio/asm v1.2.0 // indirect + github.com/shopspring/decimal v1.4.0 // indirect github.com/spiffe/go-spiffe/v2 v2.6.0 // indirect github.com/tidwall/match v1.1.1 // indirect github.com/tidwall/pretty v1.2.0 // indirect diff --git a/framework/go.sum b/framework/go.sum index 40c7bdd503b..f2df915956d 100644 --- a/framework/go.sum +++ b/framework/go.sum @@ -32,6 +32,10 @@ github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1 h1:WJ github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1/go.mod h1:tCcJZ0uHAmvjsVYzEFivsRTN00oz5BEsRgQHu5JZ9WE= github.com/AzureAD/microsoft-authentication-library-for-go v1.6.0 h1:XRzhVemXdgvJqCH0sFfrBUTnUJSBrBf7++ypk+twtRs= github.com/AzureAD/microsoft-authentication-library-for-go v1.6.0/go.mod h1:HKpQxkWaGLJ+D/5H8QRpyQXA1eKjxkFlOMwck5+33Jk= +github.com/ClickHouse/ch-go v0.61.5 h1:zwR8QbYI0tsMiEcze/uIMK+Tz1D3XZXLdNrlaOpeEI4= +github.com/ClickHouse/ch-go v0.61.5/go.mod h1:s1LJW/F/LcFs5HJnuogFMta50kKDO0lf9zzfrbl0RQg= +github.com/ClickHouse/clickhouse-go/v2 v2.30.0 h1:AG4D/hW39qa58+JHQIFOSnxyL46H6h2lrmGGk17dhFo= +github.com/ClickHouse/clickhouse-go/v2 v2.30.0/go.mod h1:i9ZQAojcayW3RsdCb3YR+n+wC2h65eJsZCscZ1Z1wyo= github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.31.0 h1:DHa2U07rk8syqvCge0QIGMCE1WxGj9njT44GH7zNJLQ= github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.31.0/go.mod h1:P4WPRUkOhJC13W//jWpyfJNDAIpvRbAUIYLX/4jtlE0= github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.55.0 h1:UnDZ/zFfG1JhH/DqxIZYU/1CUAlTUScoXD/LcM2Ykk8= @@ -125,6 +129,10 @@ github.com/felixge/httpsnoop v1.0.4 h1:NFTV2Zj1bL4mc9sqWACXbQFVBBg2W3GPvqp8/ESS2 github.com/felixge/httpsnoop v1.0.4/go.mod h1:m8KPJKqk1gH5J9DgRY2ASl2lWCfGKXixSwevea8zH2U= github.com/frankban/quicktest v1.14.6 h1:7Xjx+VpznH+oBnejlPUj8oUpdxnVs4f8XU8WnHkI4W8= github.com/frankban/quicktest v1.14.6/go.mod h1:4ptaffx2x8+WTWXmUCuVU6aPUX1/Mz7zb5vbUoiM6w0= +github.com/go-faster/city v1.0.1 h1:4WAxSZ3V2Ws4QRDrscLEDcibJY8uf41H6AhXDrNDcGw= +github.com/go-faster/city v1.0.1/go.mod h1:jKcUJId49qdW3L1qKHH/3wPeUstCVpVSXTM6vO3VcTw= +github.com/go-faster/errors v0.7.1 h1:MkJTnDoEdi9pDabt1dpWf7AA8/BaSYZqibYyhZ20AYg= +github.com/go-faster/errors v0.7.1/go.mod h1:5ySTjWFiphBs07IKuiL69nxdfd5+fzh1u7FPGZP2quo= github.com/go-jose/go-jose/v4 v4.1.4 h1:moDMcTHmvE6Groj34emNPLs/qtYXRVcd6S7NHbHz3kA= github.com/go-jose/go-jose/v4 v4.1.4/go.mod h1:x4oUasVrzR7071A4TnHLGSPpNOm2a21K9Kf04k1rs08= github.com/go-logr/logr v1.2.2/go.mod h1:jdQByPbusPIv2/zmleS9BjJVeZ6kBagPoEUsqbVz/1A= @@ -183,10 +191,15 @@ github.com/go-openapi/validate v0.25.1/go.mod h1:RMVyVFYte0gbSTaZ0N4KmTn6u/kClvA github.com/go-viper/mapstructure/v2 v2.5.0 h1:vM5IJoUAy3d7zRSVtIwQgBj7BiWtMPfmPEgAXnvj1Ro= github.com/go-viper/mapstructure/v2 v2.5.0/go.mod h1:oJDH3BJKyqBA2TXFhDsKDGDTlndYOZ6rGS0BRZIxGhM= github.com/godbus/dbus/v5 v5.0.4/go.mod h1:xhWf0FNVPg57R7Z0UbKHbJfkEywrmjJnf7w5xrFpKfA= +github.com/gogo/protobuf v1.3.2/go.mod h1:P1XiOD3dCwIKUDQYPy72D8LYyHL2YPYrpS2s69NZV8Q= github.com/golang-jwt/jwt/v5 v5.3.1 h1:kYf81DTWFe7t+1VvL7eS+jKFVWaUnK9cB1qbwn63YCY= github.com/golang-jwt/jwt/v5 v5.3.1/go.mod h1:fxCRLWMO43lRc8nhHWY6LGqRcf+1gQWArsqaEUEa5bE= +github.com/golang/protobuf v1.5.0/go.mod h1:FsONVRAS9T7sI+LIUmWTfcYkHO4aIWwzhcaSAoJOfIk= github.com/golang/protobuf v1.5.4 h1:i7eJL8qZTpSEXOPTxNKhASYpMn+8e5Q6AdndVa1dWek= github.com/golang/protobuf v1.5.4/go.mod h1:lnTiLA8Wa4RWRcIUkrtSVa5nRhsEGBg48fD6rSs7xps= +github.com/golang/snappy v0.0.1/go.mod h1:/XxbfmMg8lxefKM7IXC3fBNl/7bRcc72aCRzEWrmP2Q= +github.com/google/go-cmp v0.5.2/go.mod h1:v8dTdLbMG2kIc/vJvl+f65V22dbkXbowE6jgT/gNBxE= +github.com/google/go-cmp v0.5.5/go.mod h1:v8dTdLbMG2kIc/vJvl+f65V22dbkXbowE6jgT/gNBxE= github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8= github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU= github.com/google/martian/v3 v3.3.3 h1:DIhPTQrbPkgs2yJYdXU/eNACCG5DVQjySNRNlflZ9Fc= @@ -201,6 +214,8 @@ github.com/googleapis/gax-go/v2 v2.22.0 h1:PjIWBpgGIVKGoCXuiCoP64altEJCj3/Ei+kSU github.com/googleapis/gax-go/v2 v2.22.0/go.mod h1:irWBbALSr0Sk3qlqb9SyJ1h68WjgeFuiOzI4Rqw5+aY= github.com/hajimehoshi/go-mp3 v0.3.4 h1:NUP7pBYH8OguP4diaTZ9wJbUbk3tC0KlfzsEpWmYj68= github.com/hajimehoshi/go-mp3 v0.3.4/go.mod h1:fRtZraRFcWb0pu7ok0LqyFhCUrPeMsGRSVop0eemFmo= +github.com/hashicorp/go-version v1.6.0 h1:feTTfFNnjP967rlCxM/I9g701jU+RN74YKx2mOkIeek= +github.com/hashicorp/go-version v1.6.0/go.mod h1:fltr4n8CU8Ke44wwGCBoEymUuxUHl09ZGVZPK5anwXA= github.com/invopop/jsonschema v0.13.0 h1:KvpoAJWEjR3uD9Kbm2HWJmqsEaHt8lBUpd0qHcIi21E= github.com/invopop/jsonschema v0.13.0/go.mod h1:ffZ5Km5SWWRAIN6wbDXItl95euhFz2uON45H2qjYt+0= github.com/jackc/pgpassfile v1.0.0 h1:/6Hmqy13Ss2zCq62VdNG8tM1wchn8zjSGOBJ6icpsIM= @@ -218,12 +233,18 @@ github.com/jinzhu/now v1.1.5/go.mod h1:d3SSVoowX0Lcu0IBviAWJpolVfI5UJVZZ7cO71lE/ github.com/juju/gnuflag v0.0.0-20171113085948-2ce1bb71843d/go.mod h1:2PavIy+JPciBPrBUjwbNvtwB6RQlve+hkpll6QSNmOE= github.com/keybase/go-keychain v0.0.1 h1:way+bWYa6lDppZoZcgMbYsvC7GxljxrskdNInRtuthU= github.com/keybase/go-keychain v0.0.1/go.mod h1:PdEILRW3i9D8JcdM+FmY6RwkHGnhHxXwkPPMeUgOK1k= +github.com/kisielk/errcheck v1.5.0/go.mod h1:pFxgyoBC7bSaBwPgfKdkLd5X25qrDl4LWUI2bnpBCr8= +github.com/kisielk/gotool v1.0.0/go.mod h1:XhKaO+MFFWcvkIS/tQcRk01m1F5IRFswLeQ+oQHNcck= +github.com/klauspost/compress v1.13.6/go.mod h1:/3/Vjq9QcHkK5uEr5lBEmyoZ1iFhe47etQ6QUkpK6sk= github.com/klauspost/compress v1.18.6 h1:2jupLlAwFm95+YDR+NwD2MEfFO9d4z4Prjl1XXDjuao= github.com/klauspost/compress v1.18.6/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ= github.com/klauspost/cpuid/v2 v2.3.0 h1:S4CRMLnYUhGeDFDqkGriYKdfoFlDnMtqTiI/sFzhA9Y= github.com/klauspost/cpuid/v2 v2.3.0/go.mod h1:hqwkgyIinND0mEev00jJYCxPNVRVXFQeu1XKlok6oO0= +github.com/kr/pretty v0.1.0/go.mod h1:dAy3ld7l9f0ibDNOQOHHMYYIIbhfbHSm3C4ZsoJORNo= github.com/kr/pretty v0.3.1 h1:flRD4NNwYAUpkphVc1HcthR4KEIFJ65n8Mw5qdRn3LE= github.com/kr/pretty v0.3.1/go.mod h1:hoEshYVHaxMs3cyo3Yncou5ZscifuDolrwPKZanG3xk= +github.com/kr/pty v1.1.1/go.mod h1:pFQYn66WHrOpPYNljwOMqo10TkYh1fy3cYio2l3bCsQ= +github.com/kr/text v0.1.0/go.mod h1:4Jbv+DJW3UT/LiOwJeYQe1efqtUx/iVham/4vfdArNI= github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY= github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE= github.com/kylelemons/godebug v1.1.0 h1:RPNrshWIDI6G2gRW9EHilWtl7Z6Sb1BR0xunSBf0SNc= @@ -247,6 +268,11 @@ github.com/oapi-codegen/runtime v1.1.1 h1:EXLHh0DXIJnWhdRPN2w4MXAzFyE4CskzhNLUmt github.com/oapi-codegen/runtime v1.1.1/go.mod h1:SK9X900oXmPWilYR5/WKPzt3Kqxn/uS/+lbpREv+eCg= github.com/oklog/ulid v1.3.1 h1:EGfNDEx6MqHz8B3uNV6QAib1UR2Lm97sHi3ocA6ESJ4= github.com/oklog/ulid v1.3.1/go.mod h1:CirwcVhetQ6Lv90oh/F+FBtV6XMibvdAFo93nm5qn4U= +github.com/paulmach/orb v0.11.1 h1:3koVegMC4X/WeiXYz9iswopaTwMem53NzTJuTF20JzU= +github.com/paulmach/orb v0.11.1/go.mod h1:5mULz1xQfs3bmQm63QEJA6lNGujuRafwA5S/EnuLaLU= +github.com/paulmach/protoscan v0.2.1/go.mod h1:SpcSwydNLrxUGSDvXvO0P7g7AuhJ7lcKfDlhJCDw2gY= +github.com/pierrec/lz4/v4 v4.1.21 h1:yOVMLb6qSIDP67pl/5F7RepeKYu/VmTyEXvuMI5d9mQ= +github.com/pierrec/lz4/v4 v4.1.21/go.mod h1:gZWDp/Ze/IJXGXf23ltt2EXimqmTUXEy0GFuRQyBid4= github.com/pinecone-io/go-pinecone/v5 v5.3.0 h1:0YQlEtmXGWK/I8ztkOVM6PuBYgFJZhjSdb0ddU+bHPE= github.com/pinecone-io/go-pinecone/v5 v5.3.0/go.mod h1:6Fg85fcyvMUQFf9KW7zniN81kelSYvsjF+KPLdc1MGA= github.com/pkg/browser v0.0.0-20240102092130-5ac0b6a4141c h1:+mdjkGKdHQG3305AYmdv1U2eRNDiU2ErMBj1gwrq8eQ= @@ -269,6 +295,10 @@ github.com/rs/zerolog v1.34.0 h1:k43nTLIwcTVQAncfCw4KZ2VY6ukYoZaBPNOE8txlOeY= github.com/rs/zerolog v1.34.0/go.mod h1:bJsvje4Z08ROH4Nhs5iH600c3IkWhwp44iRc54W6wYQ= github.com/savsgio/gotils v0.0.0-20250408102913-196191ec6287 h1:qIQ0tWF9vxGtkJa24bR+2i53WBCz1nW/Pc47oVYauC4= github.com/savsgio/gotils v0.0.0-20250408102913-196191ec6287/go.mod h1:sM7Mt7uEoCeFSCBM+qBrqvEo+/9vdmj19wzp3yzUhmg= +github.com/segmentio/asm v1.2.0 h1:9BQrFxC+YOHJlTlHGkTrFWf59nbL3XnCoFLTwDCI7ys= +github.com/segmentio/asm v1.2.0/go.mod h1:BqMnlJP91P8d+4ibuonYZw9mfnzI9HfxselHZr5aAcs= +github.com/shopspring/decimal v1.4.0 h1:bxl37RwXBklmTi0C79JfXCEBD1cqqHt0bbgBAGFp81k= +github.com/shopspring/decimal v1.4.0/go.mod h1:gawqmDU56v4yIKSwfBSFip1HdCCXN8/+DMd9qYNcwME= github.com/spf13/cast v1.10.0 h1:h2x0u2shc1QuLHfxi+cTJvs30+ZAHOGRic8uyGTDWxY= github.com/spf13/cast v1.10.0/go.mod h1:jNfB8QC9IA6ZuY2ZjDp0KtFO2LZZlg4S/7bzP6qqeHo= github.com/spiffe/go-spiffe/v2 v2.6.0 h1:l+DolpxNWYgruGQVV0xsfeya3CsC7m8iBzDnMpsbLuo= @@ -281,6 +311,7 @@ github.com/stretchr/objx v0.5.2/go.mod h1:FRsXN1f5AsAjCGJKqEizvkpNtU+EGNCLh3NxZ/ github.com/stretchr/objx v0.5.3 h1:jmXUvGomnU1o3W/V5h2VEradbpJDwGrzugQQvL0POH4= github.com/stretchr/objx v0.5.3/go.mod h1:rDQraq+vQZU7Fde9LOZLr8Tax6zZvy4kuNKF+QYS+U0= github.com/stretchr/testify v1.3.0/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI= +github.com/stretchr/testify v1.6.1/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg= github.com/stretchr/testify v1.7.0/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg= github.com/stretchr/testify v1.7.1/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg= github.com/stretchr/testify v1.8.0/go.mod h1:yNjHg4UonilssWZ8iaSj1OCr/vHnekPRkoO+kdMU+MU= @@ -293,6 +324,7 @@ github.com/tidwall/gjson v1.18.0 h1:FIDeeyB800efLX89e5a8Y0BNH+LOngJyGrIWxG2FKQY= github.com/tidwall/gjson v1.18.0/go.mod h1:/wbyibRr2FHMks5tjHJ5F8dMZh3AcwJEMf5vlfC0lxk= github.com/tidwall/match v1.1.1 h1:+Ho715JplO36QYgwN9PGYNhgZvoUSc9X2c80KVTi+GA= github.com/tidwall/match v1.1.1/go.mod h1:eRSPERbgtNPcGhD8UCthc6PmLEQXEWd3PRB5JTxsfmM= +github.com/tidwall/pretty v1.0.0/go.mod h1:XNkn88O1ChpSDQmQeStsy+sBenx6DDtFZJxhVysOjyk= github.com/tidwall/pretty v1.2.0 h1:RWIZEg2iJ8/g6fDDYzMpobmaoGh5OLl4AXtGUGPcqCs= github.com/tidwall/pretty v1.2.0/go.mod h1:ITEVvHYasfjBbM0u2Pg8T2nJnzm8xPwvNhhsoaGGjNU= github.com/tidwall/sjson v1.2.5 h1:kLy8mja+1c9jlljvWTlSazM7cKDRfJuR/bOJhcY5NcY= @@ -309,10 +341,17 @@ github.com/weaviate/weaviate-go-client/v5 v5.7.1 h1:vEMxh486QqRqWaq58UEe/TiTbGbo github.com/weaviate/weaviate-go-client/v5 v5.7.1/go.mod h1:T/JDErjN074GrnYIa0AgK1TGUGP/6A/8vqXNPlv4c6E= github.com/wk8/go-ordered-map/v2 v2.1.8 h1:5h/BUHu93oj4gIdvHHHGsScSTMijfx5PeYkE/fJgbpc= github.com/wk8/go-ordered-map/v2 v2.1.8/go.mod h1:5nJHM5DyteebpVlHnWMV0rPz6Zp7+xBAnxjb1X5vnTw= +github.com/xdg-go/pbkdf2 v1.0.0/go.mod h1:jrpuAogTd400dnrH08LKmI/xc1MbPOebTwRqcT5RDeI= +github.com/xdg-go/scram v1.1.1/go.mod h1:RaEWvsqvNKKvBPvcKeFjrG2cJqOkHTiyTpzz23ni57g= +github.com/xdg-go/stringprep v1.0.3/go.mod h1:W3f5j4i+9rC0kuIEJL0ky1VpHXQU3ocBgklLGvcBnW8= github.com/xyproto/randomstring v1.0.5 h1:YtlWPoRdgMu3NZtP45drfy1GKoojuR7hmRcnhZqKjWU= github.com/xyproto/randomstring v1.0.5/go.mod h1:rgmS5DeNXLivK7YprL0pY+lTuhNQW3iGxZ18UQApw/E= github.com/yosida95/uritemplate/v3 v3.0.2 h1:Ed3Oyj9yrmi9087+NczuL5BwkIc4wvTb5zIM+UJPGz4= github.com/yosida95/uritemplate/v3 v3.0.2/go.mod h1:ILOh0sOhIJR3+L/8afwt/kE++YT040gmv5BQTMR2HP4= +github.com/youmark/pkcs8 v0.0.0-20181117223130-1be2e3e5546d/go.mod h1:rHwXgn7JulP+udvsHwJoVG1YGAP6VLg4y9I5dyZdqmA= +github.com/yuin/goldmark v1.1.27/go.mod h1:3hX8gzYuyVAZsxl0MRgGTJEmQBFcNTphYh9decYSb74= +github.com/yuin/goldmark v1.2.1/go.mod h1:3hX8gzYuyVAZsxl0MRgGTJEmQBFcNTphYh9decYSb74= +go.mongodb.org/mongo-driver v1.11.4/go.mod h1:PTSz5yu21bkT/wXpkS7WR5f0ddqw5quethTUn9WM+2g= go.mongodb.org/mongo-driver v1.17.7 h1:a9w+U3Vt67eYzcfq3k/OAv284/uUUkL0uP75VE5rCOU= go.mongodb.org/mongo-driver v1.17.7/go.mod h1:Hy04i7O2kC4RS06ZrhPRqj/u4DTYkFDAAccj+rVKqgQ= go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ64= @@ -341,24 +380,58 @@ go.yaml.in/yaml/v3 v3.0.4 h1:tfq32ie2Jv2UxXFdLJdh3jXuOzWiL1fo0bu/FbuKpbc= go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg= golang.org/x/arch v0.23.0 h1:lKF64A2jF6Zd8L0knGltUnegD62JMFBiCPBmQpToHhg= golang.org/x/arch v0.23.0/go.mod h1:dNHoOeKiyja7GTvF9NJS1l3Z2yntpQNzgrjh1cU103A= +golang.org/x/crypto v0.0.0-20190308221718-c2843e01d9a2/go.mod h1:djNgcEr1/C05ACkg1iLfiJU5Ep61QUkGW8qpdssI0+w= +golang.org/x/crypto v0.0.0-20191011191535-87dc89f01550/go.mod h1:yigFU9vqHzYiE8UmvKecakEJjdnWj3jj499lnFckfCI= +golang.org/x/crypto v0.0.0-20200622213623-75b288015ac9/go.mod h1:LzIPMQfyMNhhGPhUkYOs5KpL4U8rLKemX1yGLhDgUto= +golang.org/x/crypto v0.0.0-20220622213112-05595931fe9d/go.mod h1:IxCIyHEi3zRg3s0A5j5BB6A9Jmi73HwBIUl50j+osU4= golang.org/x/crypto v0.52.0 h1:RMs7fP2rXdep0CftQlK8Uf+kibLm7qkCcradZWYz988= golang.org/x/crypto v0.52.0/go.mod h1:1QgfPxDqh0T2M/elOJtp9RvuR95kVjir0e6/BvEmGbc= +golang.org/x/mod v0.2.0/go.mod h1:s0Qsj1ACt9ePp/hMypM3fl4fZqREWJwdYDEqhRiZZUA= +golang.org/x/mod v0.3.0/go.mod h1:s0Qsj1ACt9ePp/hMypM3fl4fZqREWJwdYDEqhRiZZUA= +golang.org/x/net v0.0.0-20190404232315-eb5bcb51f2a3/go.mod h1:t9HGtf8HONx5eT2rtn7q6eTqICYqUVnKs3thJo3Qplg= +golang.org/x/net v0.0.0-20190620200207-3b0461eec859/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s= +golang.org/x/net v0.0.0-20200226121028-0de0cce0169b/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s= +golang.org/x/net v0.0.0-20201021035429-f5854403a974/go.mod h1:sp8m0HH+o8qH0wwXwYZr8TS3Oi6o0r6Gce1SSxlDquU= +golang.org/x/net v0.0.0-20211112202133-69e39bad7dc2/go.mod h1:9nx3DQGgdP8bBQD5qxJ1jj9UTztislL4KSBs9R2vV5Y= golang.org/x/net v0.55.0 h1:bcvxaJn3e1U6InsFWt1JUq1aSjnRxLzT2rtD2KfkDF8= golang.org/x/net v0.55.0/go.mod h1:L5U2KuzuOe1lY7Z+aWVIKK6qEeJXnXV9yzGA+WCHJww= golang.org/x/oauth2 v0.36.0 h1:peZ/1z27fi9hUOFCAZaHyrpWG5lwe0RJEEEeH0ThlIs= golang.org/x/oauth2 v0.36.0/go.mod h1:YDBUJMTkDnJS+A4BP4eZBjCqtokkg1hODuPjwiGPO7Q= +golang.org/x/sync v0.0.0-20190423024810-112230192c58/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= +golang.org/x/sync v0.0.0-20190911185100-cd5d95a43a6e/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= +golang.org/x/sync v0.0.0-20201020160332-67f06af15bc9/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= +golang.org/x/sync v0.0.0-20210220032951-036812b2e83c/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= golang.org/x/sync v0.20.0 h1:e0PTpb7pjO8GAtTs2dQ6jYa5BWYlMuX047Dco/pItO4= golang.org/x/sync v0.20.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0= +golang.org/x/sys v0.0.0-20190215142949-d0b11bdaac8a/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY= +golang.org/x/sys v0.0.0-20190412213103-97732733099d/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20200930185726-fdedc70b468f/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20201119102817-f84b799fce68/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20210423082822-04245dca01da/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20210615035016-665e8c7367d1/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.0.0-20220811171246-fbc7d0a398ab/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.1.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.12.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.45.0 h1:dO4czNzziLiiXplLQgBCEpCvXQ3dnkn0SdaZSYdQ+FY= golang.org/x/sys v0.45.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= +golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo= +golang.org/x/text v0.3.0/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ= +golang.org/x/text v0.3.3/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ= +golang.org/x/text v0.3.6/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ= +golang.org/x/text v0.3.7/go.mod h1:u+2+/6zg+i71rQMx5EYifcz6MCKuco9NR6JIITiCfzQ= golang.org/x/text v0.37.0 h1:Cqjiwd9eSg8e0QAkyCaQTNHFIIzWtidPahFWR83rTrc= golang.org/x/text v0.37.0/go.mod h1:a5sjxXGs9hsn/AJVwuElvCAo9v8QYLzvavO5z2PiM38= golang.org/x/time v0.15.0 h1:bbrp8t3bGUeFOx08pvsMYRTCVSMk89u4tKbNOZbp88U= golang.org/x/time v0.15.0/go.mod h1:Y4YMaQmXwGQZoFaVFk4YpCt4FLQMYKZe9oeV/f4MSno= +golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ= +golang.org/x/tools v0.0.0-20191119224855-298f0cb1881e/go.mod h1:b+2E5dAYhXwXZwtnZ6UAqBI28+e2cm9otk0dWdXHAEo= +golang.org/x/tools v0.0.0-20200619180055-7c47624df98f/go.mod h1:EkVYQZoAsY45+roYkvgYkIh4xh/qjgUK9TdY2XT94GE= +golang.org/x/tools v0.0.0-20210106214847-113979e3529a/go.mod h1:emZCQorbCU4vsT4fOWvOPXz4eW1wZW4PmDk9uLelYpA= +golang.org/x/xerrors v0.0.0-20190717185122-a985d3407aa7/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= +golang.org/x/xerrors v0.0.0-20191011141410-1b5146add898/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= +golang.org/x/xerrors v0.0.0-20191204190536-9bdfabe68543/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= +golang.org/x/xerrors v0.0.0-20200804184101-5ec99f83aff1/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= gonum.org/v1/gonum v0.17.0 h1:VbpOemQlsSMrYmn7T2OUvQ4dqxQXU+ouZFQsZOx50z4= gonum.org/v1/gonum v0.17.0/go.mod h1:El3tOrEuMpv2UdMrbNlKEh9vd86bmQ6vqIcDwxEOc1E= google.golang.org/api v0.282.0 h1:WmJiSVqUnKqJCpJOx7YADbXaC+9DDsnGSfllFSj7R2I= @@ -371,14 +444,19 @@ google.golang.org/genproto/googleapis/rpc v0.0.0-20260523011958-0a33c5d7ca68 h1: google.golang.org/genproto/googleapis/rpc v0.0.0-20260523011958-0a33c5d7ca68/go.mod h1:4Hqkh8ycfw05ld/3BWL7rJOSfebL2Q+DVDeRgYgxUU8= google.golang.org/grpc v1.81.1 h1:VnnIIZ88UzOOKLukQi+ImGz8O1Wdp8nAGGnvOfEIWQQ= google.golang.org/grpc v1.81.1/go.mod h1:xGH9GfzOyMTGIOXBJmXt+BX/V0kcdQbdcuwQ/zNw42I= +google.golang.org/protobuf v1.26.0-rc.1/go.mod h1:jlhhOSvTdKEhbULTjvd4ARK9grFBp09yW+WbY/TyQbw= +google.golang.org/protobuf v1.27.1/go.mod h1:9q0QmTI4eRPtz6boOQmLYwt+qCgq0jsYwAQnmE0givc= google.golang.org/protobuf v1.36.12-0.20260120151049-f2248ac996af h1:+5/Sw3GsDNlEmu7TfklWKPdQ0Ykja5VEmq2i817+jbI= google.golang.org/protobuf v1.36.12-0.20260120151049-f2248ac996af/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco= gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= +gopkg.in/check.v1 v1.0.0-20180628173108-788fd7840127/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk= gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EVd6muEfDQjcINNoR0C8j2r3qZ4Q= gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= +gorm.io/driver/clickhouse v0.7.0 h1:BCrqvgONayvZRgtuA6hdya+eAW5P2QVagV3OlEp1vtA= +gorm.io/driver/clickhouse v0.7.0/go.mod h1:TmNo0wcVTsD4BBObiRnCahUgHJHjBIwuRejHwYt3JRs= gorm.io/driver/postgres v1.6.0 h1:2dxzU8xJ+ivvqTRph34QX+WrRaJlmfyPqXmoGVjMBa4= gorm.io/driver/postgres v1.6.0/go.mod h1:vUw0mrGgrTK+uPHEhAdV4sfFELrByKVGnaVRkXDhtWo= gorm.io/driver/sqlite v1.6.0 h1:WHRRrIiulaPiPFmDcod6prc4l2VGVWHz80KspNsxSfQ= diff --git a/framework/logstore/asyncjob_test.go b/framework/logstore/asyncjob_test.go index c01bdef0080..52a6c09c754 100644 --- a/framework/logstore/asyncjob_test.go +++ b/framework/logstore/asyncjob_test.go @@ -45,7 +45,7 @@ func newTestAsyncExecutor(t *testing.T) *AsyncJobExecutor { govStore := &testGovernanceStore{ virtualKeys: map[string]*configstoreTables.TableVirtualKey{ - "sk-bf-test": {ID: "vk-123", Value: "sk-bf-test"}, + "sk-bf-test": {ID: "vk-123", Value: *schemas.NewSecretVar("sk-bf-test")}, }, } diff --git a/framework/logstore/clickhouse.go b/framework/logstore/clickhouse.go new file mode 100644 index 00000000000..5080c7a2f8b --- /dev/null +++ b/framework/logstore/clickhouse.go @@ -0,0 +1,196 @@ +package logstore + +import ( + "context" + "fmt" + "net" + "net/url" + "strings" + "time" + + "github.com/maximhq/bifrost/core/schemas" + clickhousedriver "gorm.io/driver/clickhouse" + "gorm.io/gorm" +) + +// ClickHouseConfig represents the configuration for a ClickHouse log store. +// +// ClickHouse is an append-only columnar OLAP store. The backend uses +// ReplacingMergeTree tables with a connection-level `final = 1` setting so +// reads transparently see the latest version of each row (see clickhousestore.go +// for the mutation strategy). +type ClickHouseConfig struct { + Host *schemas.SecretVar `json:"host"` + Port *schemas.SecretVar `json:"port"` + Database *schemas.SecretVar `json:"database"` + Username *schemas.SecretVar `json:"username"` + Password *schemas.SecretVar `json:"password"` + // Protocol selects the ClickHouse wire protocol: "native" (default, port + // 9000/9440) or "http" (port 8123/8443). clickhouse-go derives the protocol + // from the DSN scheme, so this maps to clickhouse:// vs http(s)://. + Protocol string `json:"protocol,omitempty"` + // Secure enables TLS (native: secure=true; http: switches to https). + Secure bool `json:"secure,omitempty"` + // DialTimeout is the connection dial timeout in milliseconds (JSON config + // duration fields are integer milliseconds). 0 means the 10s default. + DialTimeout int `json:"dial_timeout,omitempty"` + // Cluster, when set, makes DDL run as `ON CLUSTER ` against + // ReplicatedReplacingMergeTree engines. Empty means single-node. + Cluster string `json:"cluster,omitempty"` +} + +const ( + defaultClickHouseNativePort = "9000" + defaultClickHouseNativeTLSPort = "9440" + defaultClickHouseHTTPPort = "8123" + defaultClickHouseHTTPSPort = "8443" + defaultClickHouseDatabase = "default" + defaultClickHouseDialTimeout = 10 * time.Second + clickHouseProtocolNative = "native" + clickHouseProtocolHTTP = "http" +) + +func secretValue(v *schemas.SecretVar) string { + if v == nil { + return "" + } + return v.GetValue() +} + +// buildClickHouseDSN assembles a clickhouse-go v2 DSN. The wire protocol is +// selected via the URL scheme (clickhouse:// = native, http(s):// = HTTP), and +// unknown query params (here, `final`) are passed through as ClickHouse +// settings, so every pooled connection applies FINAL automatically. +func buildClickHouseDSN(config *ClickHouseConfig) (string, error) { + host := secretValue(config.Host) + if host == "" { + return "", fmt.Errorf("clickhouse: host is required") + } + + // Resolve protocol -> URL scheme + default port. clickhouse-go requires + // scheme "https" (not "http" + secure) for HTTP-over-TLS. + var scheme, defaultPort string + switch strings.ToLower(strings.TrimSpace(config.Protocol)) { + case "", clickHouseProtocolNative: + scheme = "clickhouse" + if config.Secure { + defaultPort = defaultClickHouseNativeTLSPort + } else { + defaultPort = defaultClickHouseNativePort + } + case clickHouseProtocolHTTP: + if config.Secure { + scheme = "https" + defaultPort = defaultClickHouseHTTPSPort + } else { + scheme = "http" + defaultPort = defaultClickHouseHTTPPort + } + default: + return "", fmt.Errorf("clickhouse: unsupported protocol %q (use %q or %q)", config.Protocol, clickHouseProtocolNative, clickHouseProtocolHTTP) + } + + port := secretValue(config.Port) + if port == "" { + port = defaultPort + } + + database := secretValue(config.Database) + if database == "" { + database = defaultClickHouseDatabase + } + + dialTimeout := defaultClickHouseDialTimeout + if config.DialTimeout > 0 { + dialTimeout = time.Duration(config.DialTimeout) * time.Millisecond + } + + u := url.URL{ + Scheme: scheme, + Host: net.JoinHostPort(host, port), + Path: "/" + database, + } + if user := secretValue(config.Username); user != "" { + if pass := secretValue(config.Password); pass != "" { + u.User = url.UserPassword(user, pass) + } else { + u.User = url.User(user) + } + } + + q := url.Values{} + // Apply FINAL to every query so ReplacingMergeTree dedup is transparent to + // the reused analytics read path (see clickhousestore.go). + q.Set("final", "1") + // The GORM ClickHouse driver rewrites DELETE/UPDATE into ALTER TABLE + // mutations, which are asynchronous by default - a read right after a + // delete would still see the rows. mutations_sync=1 makes the connection + // wait until the mutation is applied on the current replica. + q.Set("mutations_sync", "1") + // The shared analytics SQL aliases aggregates with column names + // (SUM(cost) AS cost) while filters reference the same names in WHERE. + // ClickHouse resolves identifiers in WHERE to SELECT aliases by default + // (error 184: aggregate function found in WHERE); this setting restores + // the standard-SQL column-first resolution Postgres/SQLite use. + q.Set("prefer_column_name_to_alias", "1") + q.Set("dial_timeout", dialTimeout.String()) + // clickhouse-go: native TLS is requested via secure=true; the https scheme + // also requires secure=true; plain http must NOT set it. + if config.Secure { + q.Set("secure", "true") + } + u.RawQuery = q.Encode() + + return u.String(), nil +} + +// newClickHouseLogStore creates a new ClickHouse log store. retentionDays drives +// the table TTL; values < 1 leave TTL unset (the LogsCleaner still prunes via +// DeleteLogsBatch). +func newClickHouseLogStore(ctx context.Context, config *ClickHouseConfig, retentionDays int, logger schemas.Logger) (LogStore, error) { + dsn, err := buildClickHouseDSN(config) + if err != nil { + return nil, err + } + + logger.Info("logstore: opening clickhouse connection (if this step hangs, the database host/port is likely unreachable)") + db, err := gorm.Open(clickhousedriver.Open(dsn), &gorm.Config{ + Logger: newGormLogger(logger), + }) + if err != nil { + logger.Error("logstore: failed to open clickhouse connection: %v", err) + return nil, err + } + + // Release the pool on any startup failure past this point; ownership + // transfers to the returned store only on success. + constructed := false + defer func() { + if constructed { + return + } + if sqlDB, dbErr := db.DB(); dbErr == nil { + if closeErr := sqlDB.Close(); closeErr != nil { + logger.Error("logstore: failed to close clickhouse pool after startup failure: %v", closeErr) + } + } + }() + + if err := db.WithContext(ctx).Exec("SELECT 1").Error; err != nil { + logger.Error("logstore: clickhouse ping failed: %v", err) + return nil, fmt.Errorf("clickhouse ping failed: %w", err) + } + + logger.Info("logstore: running clickhouse schema migrations") + if err := triggerClickHouseMigrations(ctx, db, config.Cluster, retentionDays, logger); err != nil { + logger.Error("logstore: clickhouse schema migrations failed: %v", err) + return nil, err + } + logger.Info("logstore: clickhouse schema migrations complete") + + constructed = true + return &ClickHouseLogStore{ + RDBLogStore: &RDBLogStore{db: db, logger: logger}, + cluster: config.Cluster, + }, nil +} diff --git a/framework/logstore/clickhousemigrate.go b/framework/logstore/clickhousemigrate.go new file mode 100644 index 00000000000..f51641c3e35 --- /dev/null +++ b/framework/logstore/clickhousemigrate.go @@ -0,0 +1,289 @@ +package logstore + +import ( + "context" + "fmt" + "reflect" + "strings" + "time" + + "github.com/maximhq/bifrost/core/schemas" + "gorm.io/gorm" + "gorm.io/gorm/schema" +) + +// clickhouseColumnType maps a GORM-parsed field to a ClickHouse column type. +// Pointer fields become Nullable(...). This keeps the ClickHouse DDL in lockstep +// with the shared Log/MCPToolLog/AsyncJob structs so the reused read path (which +// references DB column names) never drifts from the physical schema. +func clickhouseColumnType(f *schema.Field) string { + ft := f.FieldType + nullable := false + for ft.Kind() == reflect.Ptr { + nullable = true + ft = ft.Elem() + } + + base := "String" + switch ft.Kind() { + case reflect.String: + base = "String" + case reflect.Bool: + base = "Bool" + case reflect.Int, reflect.Int8, reflect.Int16, reflect.Int32, reflect.Int64: + base = "Int64" + case reflect.Uint, reflect.Uint8, reflect.Uint16, reflect.Uint32, reflect.Uint64: + base = "UInt64" + case reflect.Float32, reflect.Float64: + base = "Float64" + case reflect.Struct: + if ft == reflect.TypeFor[time.Time]() { + base = "DateTime64(3)" + } + } + + if nullable { + return "Nullable(" + base + ")" + } + return base +} + +// chColumnOverrides maps column names to full ClickHouse column definitions +// that replace the default type derived from the Go struct. Used for columns +// that need DEFAULT expressions computed by ClickHouse at INSERT time. +var chColumnOverrides = map[string]string{ + // inc_number: monotonically increasing per-row insert-order number. + // generateSnowflakeID() produces a unique UInt64 for every row in a batch + // (12-bit counter = 4096/ms), cast to Int64 for Go compatibility. + // The DEFAULT fires on fresh inserts (where the column is omitted) and is + // preserved on re-inserts (updates) because the existing value is carried + // through the read-modify-write cycle. + "inc_number": "`inc_number` Int64 DEFAULT CAST(generateSnowflakeID() AS Int64)", +} + +// clickhouseColumnDefs parses the GORM schema for model and returns column +// definitions ("`name` Type") for every persisted field, in struct order. +func clickhouseColumnDefs(db *gorm.DB, model any) ([]string, error) { + st, err := schema.Parse(model, &chSchemaCache, db.NamingStrategy) + if err != nil { + return nil, fmt.Errorf("failed to parse schema for clickhouse DDL: %w", err) + } + var cols []string + for _, f := range st.Fields { + if f.DBName == "" || f.IgnoreMigration { + continue + } + if override, ok := chColumnOverrides[f.DBName]; ok { + cols = append(cols, override) + continue + } + cols = append(cols, fmt.Sprintf("`%s` %s", f.DBName, clickhouseColumnType(f))) + } + return cols, nil +} + +// chEscapeIdentifier escapes a value for embedding inside a backtick-quoted +// ClickHouse identifier (backticks are escaped by doubling). Used for the +// config-supplied cluster name, which reaches DDL via fmt.Sprintf. +func chEscapeIdentifier(s string) string { + return strings.ReplaceAll(s, "`", "``") +} + +// chTableOpts describes the engine-level options for a ClickHouse table. +type chTableOpts struct { + table string + partitionBy string // empty = no PARTITION BY + orderBy string // e.g. "(timestamp, id)" + ttl string // empty = no TTL + skipIndexes []string // full "INDEX ..." clauses +} + +// clickhouseCreateTable derives the column list from the GORM model, appends the +// ReplacingMergeTree version column (`ver`, defaulted to now64() so every INSERT +// is auto-versioned), and runs an idempotent CREATE TABLE IF NOT EXISTS. +func clickhouseCreateTable(ctx context.Context, db *gorm.DB, model any, opts chTableOpts, cluster string) error { + cols, err := clickhouseColumnDefs(db, model) + if err != nil { + return err + } + // Version column for ReplacingMergeTree dedup. Not part of the Go struct, so + // INSERTs omit it and ClickHouse fills now64(); a later re-insert (cost + // backfill, has_object flip, idempotent retry) gets a higher ver and wins. + // Nanosecond precision: with now64()'s default millisecond resolution, a + // create + immediate update landing in the same millisecond would tie on + // `ver` and leave the winner to merge order instead of latest-write-wins. + cols = append(cols, "`ver` DateTime64(9) DEFAULT now64(9)") + cols = append(cols, opts.skipIndexes...) + + engine := "ReplacingMergeTree(ver)" + onCluster := "" + if cluster != "" { + onCluster = fmt.Sprintf(" ON CLUSTER `%s`", chEscapeIdentifier(cluster)) + engine = fmt.Sprintf("ReplicatedReplacingMergeTree('/clickhouse/tables/{shard}/%s', '{replica}', ver)", opts.table) + } + + var b strings.Builder + fmt.Fprintf(&b, "CREATE TABLE IF NOT EXISTS `%s`%s (\n %s\n) ENGINE = %s\n", opts.table, onCluster, strings.Join(cols, ",\n "), engine) + if opts.partitionBy != "" { + fmt.Fprintf(&b, "PARTITION BY %s\n", opts.partitionBy) + } + fmt.Fprintf(&b, "ORDER BY %s\n", opts.orderBy) + if opts.ttl != "" { + fmt.Fprintf(&b, "TTL %s\n", opts.ttl) + } + b.WriteString("SETTINGS index_granularity = 8192") + + return db.WithContext(ctx).Exec(b.String()).Error +} + +// clickhouseExistingColumns returns the set of column names already present on a +// ClickHouse table. +func clickhouseExistingColumns(ctx context.Context, db *gorm.DB, table string) (map[string]struct{}, error) { + var names []string + if err := db.WithContext(ctx). + Raw("SELECT name FROM system.columns WHERE database = currentDatabase() AND table = ?", table). + Scan(&names).Error; err != nil { + return nil, err + } + set := make(map[string]struct{}, len(names)) + for _, n := range names { + set[n] = struct{}{} + } + return set, nil +} + +// clickhouseReconcileColumns adds any model columns missing from the live table +// via ALTER TABLE ... ADD COLUMN IF NOT EXISTS. This is the forward-evolution +// path: the shared Log/MCPToolLog structs gain fields over time, and CREATE +// TABLE IF NOT EXISTS only ever runs once. We do this ourselves (rather than the +// gorm driver's AutoMigrate) because AutoMigrate maps the structs' Postgres/ +// SQLite tags (type:varchar(255) -> FixedString(255), etc.) incorrectly for +// ClickHouse; clickhouseColumnType maps Go kinds to clean String/Nullable types. +func clickhouseReconcileColumns(ctx context.Context, db *gorm.DB, model any, table, cluster string, logger schemas.Logger) error { + existing, err := clickhouseExistingColumns(ctx, db, table) + if err != nil { + return fmt.Errorf("clickhouse: read columns for %s: %w", table, err) + } + st, err := schema.Parse(model, &chSchemaCache, db.NamingStrategy) + if err != nil { + return fmt.Errorf("clickhouse: parse schema for %s: %w", table, err) + } + onCluster := "" + if cluster != "" { + onCluster = fmt.Sprintf(" ON CLUSTER `%s`", chEscapeIdentifier(cluster)) + } + for _, f := range st.Fields { + if f.DBName == "" || f.IgnoreMigration { + continue + } + if _, ok := existing[f.DBName]; ok { + continue + } + columnDef := fmt.Sprintf("`%s` %s", f.DBName, clickhouseColumnType(f)) + if override, ok := chColumnOverrides[f.DBName]; ok { + columnDef = override + } + stmt := fmt.Sprintf("ALTER TABLE `%s`%s ADD COLUMN IF NOT EXISTS %s", table, onCluster, columnDef) + logger.Info("[logstore] clickhouse: adding column %s.%s", table, f.DBName) + if err := db.WithContext(ctx).Exec(stmt).Error; err != nil { + return fmt.Errorf("clickhouse: add column %s.%s: %w", table, f.DBName, err) + } + } + return nil +} + +// chLogsTTL derives the logs/mcp_tool_logs TTL clause from the configured +// retention. Values < 1 leave TTL unset (the LogsCleaner still prunes via +// DeleteLogsBatch). +func chLogsTTL(retentionDays int) string { + if retentionDays < 1 { + return "" + } + return fmt.Sprintf("toDateTime(created_at) + INTERVAL %d DAY", retentionDays) +} + +// clickhouseMigrationStep is one per-table migration: create the table if +// missing, then reconcile any columns added to its model since. +type clickhouseMigrationStep func(ctx context.Context, db *gorm.DB, cluster string, retentionDays int, logger schemas.Logger) error + +// migrationClickHouseLogsTable creates the logs table and reconciles it with +// the Log struct. +func migrationClickHouseLogsTable(ctx context.Context, db *gorm.DB, cluster string, retentionDays int, logger schemas.Logger) error { + logger.Info("[logstore] clickhouse: creating table logs") + if err := clickhouseCreateTable(ctx, db, &Log{}, chTableOpts{ + table: "logs", + partitionBy: "toYYYYMM(timestamp)", + orderBy: "(timestamp, id)", + ttl: chLogsTTL(retentionDays), + skipIndexes: []string{ + "INDEX idx_logs_provider provider TYPE bloom_filter GRANULARITY 1", + "INDEX idx_logs_model model TYPE bloom_filter GRANULARITY 1", + "INDEX idx_logs_status status TYPE bloom_filter GRANULARITY 1", + "INDEX idx_logs_team_id team_id TYPE bloom_filter GRANULARITY 1", + "INDEX idx_logs_virtual_key_id virtual_key_id TYPE bloom_filter GRANULARITY 1", + "INDEX idx_logs_user_id user_id TYPE bloom_filter GRANULARITY 1", + "INDEX idx_logs_selected_key_id selected_key_id TYPE bloom_filter GRANULARITY 1", + }, + }, cluster); err != nil { + return fmt.Errorf("clickhouse: create logs table: %w", err) + } + return clickhouseReconcileColumns(ctx, db, &Log{}, "logs", cluster, logger) +} + +// migrationClickHouseMCPToolLogsTable creates the mcp_tool_logs table and +// reconciles it with the MCPToolLog struct. +func migrationClickHouseMCPToolLogsTable(ctx context.Context, db *gorm.DB, cluster string, retentionDays int, logger schemas.Logger) error { + logger.Info("[logstore] clickhouse: creating table mcp_tool_logs") + if err := clickhouseCreateTable(ctx, db, &MCPToolLog{}, chTableOpts{ + table: "mcp_tool_logs", + partitionBy: "toYYYYMM(timestamp)", + orderBy: "(timestamp, id)", + ttl: chLogsTTL(retentionDays), + skipIndexes: []string{ + "INDEX idx_mcp_logs_status status TYPE bloom_filter GRANULARITY 1", + "INDEX idx_mcp_logs_virtual_key_id virtual_key_id TYPE bloom_filter GRANULARITY 1", + "INDEX idx_mcp_logs_tool_name tool_name TYPE bloom_filter GRANULARITY 1", + }, + }, cluster); err != nil { + return fmt.Errorf("clickhouse: create mcp_tool_logs table: %w", err) + } + return clickhouseReconcileColumns(ctx, db, &MCPToolLog{}, "mcp_tool_logs", cluster, logger) +} + +// migrationClickHouseAsyncJobsTable creates the async_jobs table and reconciles +// it with the AsyncJob struct. async_jobs is a small queue: a hard 7-day TTL on +// created_at is a safety backstop (independent of the logs retention setting); +// the AsyncJobCleaner's DeleteExpired/DeleteStale handle normal (sub-hour) +// expiry. +func migrationClickHouseAsyncJobsTable(ctx context.Context, db *gorm.DB, cluster string, _ int, logger schemas.Logger) error { + logger.Info("[logstore] clickhouse: creating table async_jobs") + if err := clickhouseCreateTable(ctx, db, &AsyncJob{}, chTableOpts{ + table: "async_jobs", + orderBy: "id", + ttl: "toDateTime(created_at) + INTERVAL 7 DAY", + }, cluster); err != nil { + return fmt.Errorf("clickhouse: create async_jobs table: %w", err) + } + return clickhouseReconcileColumns(ctx, db, &AsyncJob{}, "async_jobs", cluster, logger) +} + +// clickhouseMigrationSteps lists the per-table migrations in execution order, +// mirroring logstoreMigrationSteps for the SQL stores. +var clickhouseMigrationSteps = []clickhouseMigrationStep{ + migrationClickHouseLogsTable, + migrationClickHouseMCPToolLogsTable, + migrationClickHouseAsyncJobsTable, +} + +// triggerClickHouseMigrations runs all registered ClickHouse table migrations +// in order. Analogous to triggerMigrations for Postgres/SQLite, but with no +// migration ledger or advisory lock: CREATE TABLE IF NOT EXISTS and ADD COLUMN +// IF NOT EXISTS are inherently idempotent and concurrency-safe across pods. +func triggerClickHouseMigrations(ctx context.Context, db *gorm.DB, cluster string, retentionDays int, logger schemas.Logger) error { + for _, step := range clickhouseMigrationSteps { + if err := step(ctx, db, cluster, retentionDays, logger); err != nil { + return err + } + } + return nil +} diff --git a/framework/logstore/clickhousestore.go b/framework/logstore/clickhousestore.go new file mode 100644 index 00000000000..e44a05bab6f --- /dev/null +++ b/framework/logstore/clickhousestore.go @@ -0,0 +1,522 @@ +package logstore + +import ( + "context" + "errors" + "fmt" + "hash/fnv" + "reflect" + "sort" + "sync" + "time" + + "gorm.io/gorm" + "gorm.io/gorm/schema" +) + +// ClickHouseLogStore is a LogStore backed by ClickHouse. It embeds *RDBLogStore +// to reuse the (dialect-aware) analytics/read path and overrides only the +// methods ClickHouse cannot satisfy through plain GORM: +// +// - inserts that relied on ON CONFLICT DO NOTHING (ClickHouse has no upsert; +// idempotency comes from ReplacingMergeTree dedup + the connection-level +// `final = 1` setting, so a plain INSERT is correct), +// - row updates (ClickHouse has no cheap UPDATE; we read-modify-write and +// re-insert, letting the `ver` DEFAULT now64() column make the newest +// insert win on merge - see clickhousemigrate.go). +// +// Deletes are left to the embedded methods: the GORM ClickHouse driver emits +// lightweight `DELETE ... WHERE`, and TTL is the primary retention mechanism. +type ClickHouseLogStore struct { + *RDBLogStore + // cluster is the optional ON CLUSTER name (empty = single-node). Retained + // for future cluster-aware DDL. + cluster string + // rmwLocks serializes read-modify-write cycles per row key within this + // process. Because updates re-insert the whole row, two concurrent updaters + // of the same id (e.g. object offload setting has_object while the + // completion writer sets status/cost) would otherwise both read the same + // base row and the higher `ver` would silently drop the other's patch. + // Cross-pod races are not covered, but a given request id is only mutated + // by the pod that processed it. + rmwLocks [chRMWShards]sync.Mutex +} + +// chRMWShards is the number of RMW lock shards; keys are hashed onto them. +const chRMWShards = 128 + +func chRMWShard(table, id string) int { + h := fnv.New32a() + h.Write([]byte(table)) + h.Write([]byte{0}) + h.Write([]byte(id)) + return int(h.Sum32() % chRMWShards) +} + +// lockRMW locks the shard for a single row key and returns the unlock func. +func (s *ClickHouseLogStore) lockRMW(table, id string) func() { + mu := &s.rmwLocks[chRMWShard(table, id)] + mu.Lock() + return mu.Unlock +} + +// lockRMWBatch locks the distinct shards covering a set of row keys in +// ascending shard order (so concurrent batch lockers cannot deadlock) and +// returns the unlock func. +func (s *ClickHouseLogStore) lockRMWBatch(table string, ids []string) func() { + seen := make(map[int]struct{}, len(ids)) + for _, id := range ids { + seen[chRMWShard(table, id)] = struct{}{} + } + shards := make([]int, 0, len(seen)) + for sh := range seen { + shards = append(shards, sh) + } + sort.Ints(shards) + for _, sh := range shards { + s.rmwLocks[sh].Lock() + } + return func() { + for i := len(shards) - 1; i >= 0; i-- { + s.rmwLocks[shards[i]].Unlock() + } + } +} + +// chSchemaCache is a shared GORM schema parse cache reused across RMW calls. +var chSchemaCache sync.Map + +func chParseSchema(db *gorm.DB, model interface{}) (*schema.Schema, error) { + return schema.Parse(model, &chSchemaCache, db.NamingStrategy) +} + +// chImmutableColumns are the ReplacingMergeTree dedup key columns (the tables' +// ORDER BY is `(timestamp, id)` / `id` - see clickhousemigrate.go). Updates must +// never rewrite them: a reinserted row with a different key value would be a +// new logical row instead of replacing the old one, so the helpers below skip +// them the same way a SQL UPDATE never rewrites its WHERE key. +var chImmutableColumns = map[string]struct{}{ + "id": {}, + "timestamp": {}, + "inc_number": {}, // DB-assigned monotonic insert-order number; must survive re-inserts +} + +// chApplyUpdateMap applies a column->value map onto a struct pointer using the +// GORM schema field setters (which handle pointer / typed conversions). +// Dedup key columns are skipped. +func chApplyUpdateMap(ctx context.Context, st *schema.Schema, dest reflect.Value, updates map[string]interface{}) error { + for col, val := range updates { + if _, immutable := chImmutableColumns[col]; immutable { + continue + } + f, ok := st.FieldsByDBName[col] + if !ok { + continue + } + if err := f.Set(ctx, dest, val); err != nil { + return fmt.Errorf("clickhouse: set column %s: %w", col, err) + } + } + return nil +} + +// chApplyStructUpdate overlays the non-zero fields of src onto dest, mirroring +// GORM's Updates(struct) semantics (zero-valued fields are not written). +// Dedup key columns are skipped. +func chApplyStructUpdate(ctx context.Context, st *schema.Schema, dest, src reflect.Value) error { + for _, f := range st.Fields { + if f.DBName == "" { + continue + } + if _, immutable := chImmutableColumns[f.DBName]; immutable { + continue + } + val, isZero := f.ValueOf(ctx, src) + if isZero { + continue + } + if err := f.Set(ctx, dest, val); err != nil { + return fmt.Errorf("clickhouse: set column %s: %w", f.DBName, err) + } + } + return nil +} + +// chReinsert re-inserts a (possibly patched) row with hooks skipped so the +// BeforeCreate serialization does not clobber already-serialized base columns. +// The omitted `ver` column defaults to now64(), so this insert supersedes the +// prior version on the next ReplacingMergeTree merge (and immediately under +// `final = 1` reads). +func (s *ClickHouseLogStore) chReinsert(ctx context.Context, v interface{}) error { + return s.db.WithContext(ctx).Session(&gorm.Session{SkipHooks: true}).Create(v).Error +} + +// --- Inserts (existence check first; RMT dedup is last-write-wins) --- + +// ReplacingMergeTree keeps the row with the HIGHEST `ver`, so a duplicate +// INSERT would *replace* the existing row instead of being a no-op like the +// SQL stores' ON CONFLICT DO NOTHING. That inverts CreateIfNotExists +// semantics: a retried "processing" insert arriving after the completion +// update (and after the hybrid store's has_object flip) would resurrect the +// stale row and silently drop status/cost/has_object. The methods below +// therefore check existence under the RMW shard locks and insert only rows +// whose id is not already present. + +// chFilterMissing returns the entries whose id is not present in table, +// skipping nil entries and duplicate ids within the batch (first occurrence +// wins, matching ON CONFLICT DO NOTHING). Must be called under the RMW locks +// covering ids so a concurrent Update re-insert cannot interleave. +// +// When every entry carries a non-zero timestamp, the lookup is bounded to the +// batch's [min, max] timestamp range so it prunes granules via the +// (timestamp, id) primary key instead of scanning the id column. This is safe +// because retried creates reuse the original entry (same timestamp), and the +// tables' dedup key is (timestamp, id) anyway - a same-id row at a different +// timestamp would be a distinct logical row regardless of this check. +func chFilterMissing[T any](ctx context.Context, db *gorm.DB, table string, entries []*T, idOf func(*T) string, tsOf func(*T) time.Time) ([]*T, error) { + ids := make([]string, 0, len(entries)) + var minTS, maxTS time.Time + boundable := true + for _, e := range entries { + if e == nil { + continue + } + ids = append(ids, idOf(e)) + ts := tsOf(e) + if ts.IsZero() { + boundable = false + continue + } + if minTS.IsZero() || ts.Before(minTS) { + minTS = ts + } + if maxTS.IsZero() || ts.After(maxTS) { + maxTS = ts + } + } + if len(ids) == 0 { + return nil, nil + } + q := db.WithContext(ctx).Table(table).Where("id IN ?", ids) + if boundable { + // Bind epoch millis, not time.Time: the GORM ClickHouse driver formats + // time args as toDateTime('...') at SECONDS precision, silently dropping + // the sub-second part - a BETWEEN on the raw values would miss every row + // whose DateTime64(3) timestamp has a non-zero millisecond component. + q = q.Where("timestamp BETWEEN fromUnixTimestamp64Milli(?) AND fromUnixTimestamp64Milli(?)", minTS.UnixMilli(), maxTS.UnixMilli()) + } + var existing []string + if err := q.Pluck("id", &existing).Error; err != nil { + return nil, err + } + seen := make(map[string]struct{}, len(existing)) + for _, id := range existing { + seen[id] = struct{}{} + } + missing := make([]*T, 0, len(entries)) + for _, e := range entries { + if e == nil { + continue + } + id := idOf(e) + if _, ok := seen[id]; ok { + continue + } + seen[id] = struct{}{} + missing = append(missing, e) + } + return missing, nil +} + +// CreateIfNotExists inserts a log entry only when no row with the same id +// exists (see the semantics note above). +func (s *ClickHouseLogStore) CreateIfNotExists(ctx context.Context, entry *Log) error { + if entry == nil { + return fmt.Errorf("log entry is nil") + } + defer s.lockRMW("logs", entry.ID)() + missing, err := chFilterMissing(ctx, s.db, "logs", []*Log{entry}, func(l *Log) string { return l.ID }, func(l *Log) time.Time { return l.Timestamp }) + if err != nil { + return err + } + if len(missing) == 0 { + return nil + } + // Omit inc_number so ClickHouse's DEFAULT generateSnowflakeID() fires. + return s.db.WithContext(ctx).Omit("inc_number").Create(entry).Error +} + +// BatchCreateIfNotExists inserts the log entries whose ids are not already +// present. See CreateIfNotExists. +func (s *ClickHouseLogStore) BatchCreateIfNotExists(ctx context.Context, entries []*Log) error { + if len(entries) == 0 { + return nil + } + ids := make([]string, 0, len(entries)) + for _, e := range entries { + if e != nil { + ids = append(ids, e.ID) + } + } + defer s.lockRMWBatch("logs", ids)() + missing, err := chFilterMissing(ctx, s.db, "logs", entries, func(l *Log) string { return l.ID }, func(l *Log) time.Time { return l.Timestamp }) + if err != nil { + return err + } + if len(missing) == 0 { + return nil + } + // Omit inc_number so ClickHouse's DEFAULT generateSnowflakeID() fires. + return s.db.WithContext(ctx).Omit("inc_number").Create(&missing).Error +} + +// BatchCreateMCPToolLogsIfNotExists inserts the MCP tool log entries whose +// ids are not already present. See CreateIfNotExists. +func (s *ClickHouseLogStore) BatchCreateMCPToolLogsIfNotExists(ctx context.Context, entries []*MCPToolLog) error { + if len(entries) == 0 { + return nil + } + ids := make([]string, 0, len(entries)) + for _, e := range entries { + if e != nil { + ids = append(ids, e.ID) + } + } + defer s.lockRMWBatch("mcp_tool_logs", ids)() + missing, err := chFilterMissing(ctx, s.db, "mcp_tool_logs", entries, func(l *MCPToolLog) string { return l.ID }, func(l *MCPToolLog) time.Time { return l.Timestamp }) + if err != nil { + return err + } + if len(missing) == 0 { + return nil + } + // Omit inc_number so ClickHouse's DEFAULT generateSnowflakeID() fires. + return s.db.WithContext(ctx).Omit("inc_number").Create(&missing).Error +} + +// --- Updates (read-modify-write + re-insert) --- + +// Update applies an update (a column->value map, or a *Log/Log whose non-zero +// fields are written) to the log row by re-inserting a patched copy. +func (s *ClickHouseLogStore) Update(ctx context.Context, id string, entry any) error { + st, err := chParseSchema(s.db, &Log{}) + if err != nil { + return err + } + defer s.lockRMW("logs", id)() + var existing Log + if err := s.db.WithContext(ctx).Where("id = ?", id).First(&existing).Error; err != nil { + if errors.Is(err, gorm.ErrRecordNotFound) { + return ErrNotFound + } + return err + } + dest := reflect.ValueOf(&existing).Elem() + switch v := entry.(type) { + case map[string]interface{}: + if err := chApplyUpdateMap(ctx, st, dest, v); err != nil { + return err + } + case *Log: + if v == nil { + return fmt.Errorf("clickhouse: nil *Log update") + } + if err := v.SerializeFields(); err != nil { + return err + } + if err := chApplyStructUpdate(ctx, st, dest, reflect.ValueOf(v).Elem()); err != nil { + return err + } + case Log: + if err := v.SerializeFields(); err != nil { + return err + } + if err := chApplyStructUpdate(ctx, st, dest, reflect.ValueOf(&v).Elem()); err != nil { + return err + } + default: + return fmt.Errorf("clickhouse: unsupported Update entry type %T", entry) + } + return s.chReinsert(ctx, &existing) +} + +// BulkUpdateCost backfills costs by reading each chunk of rows, patching cost, +// and re-inserting. Reading the full row is required because the re-insert must +// reproduce every column (the ReplacingMergeTree dedup key includes timestamp). +func (s *ClickHouseLogStore) BulkUpdateCost(ctx context.Context, updates map[string]float64) error { + if len(updates) == 0 { + return nil + } + ids := make([]string, 0, len(updates)) + for id := range updates { + ids = append(ids, id) + } + for start := 0; start < len(ids); start += bulkUpdateCostChunkSize { + end := start + bulkUpdateCostChunkSize + if end > len(ids) { + end = len(ids) + } + chunk := ids[start:end] + if err := func() error { + defer s.lockRMWBatch("logs", chunk)() + var rows []*Log + if err := s.db.WithContext(ctx).Where("id IN ?", chunk).Find(&rows).Error; err != nil { + return err + } + if len(rows) == 0 { + return nil + } + for _, r := range rows { + cost := updates[r.ID] + r.Cost = &cost + } + return s.chReinsert(ctx, &rows) + }(); err != nil { + return err + } + } + return nil +} + +// UpdateMCPToolLog applies an update to an MCP tool log row via read-modify-write. +func (s *ClickHouseLogStore) UpdateMCPToolLog(ctx context.Context, id string, entry any) error { + st, err := chParseSchema(s.db, &MCPToolLog{}) + if err != nil { + return err + } + defer s.lockRMW("mcp_tool_logs", id)() + var existing MCPToolLog + if err := s.db.WithContext(ctx).Where("id = ?", id).First(&existing).Error; err != nil { + if errors.Is(err, gorm.ErrRecordNotFound) { + return ErrNotFound + } + return err + } + dest := reflect.ValueOf(&existing).Elem() + switch v := entry.(type) { + case map[string]interface{}: + if err := chApplyUpdateMap(ctx, st, dest, v); err != nil { + return err + } + case *MCPToolLog: + if v == nil { + return fmt.Errorf("clickhouse: nil *MCPToolLog update") + } + if err := v.SerializeFields(); err != nil { + return err + } + if err := chApplyStructUpdate(ctx, st, dest, reflect.ValueOf(v).Elem()); err != nil { + return err + } + case MCPToolLog: + if err := v.SerializeFields(); err != nil { + return err + } + if err := chApplyStructUpdate(ctx, st, dest, reflect.ValueOf(&v).Elem()); err != nil { + return err + } + default: + return fmt.Errorf("clickhouse: unsupported UpdateMCPToolLog entry type %T", entry) + } + return s.chReinsert(ctx, &existing) +} + +// DeleteLogsBatch deletes logs older than cutoff in batches. Overridden +// because the GORM ClickHouse driver rewrites DELETE into an ALTER TABLE +// mutation whose driver result reports 0 rows affected - the inherited +// implementation would always return 0 and the LogsCleaner would treat every +// batch as empty and stop early. The ids are selected first, so their count +// is the deleted count once the (mutations_sync=1) delete returns. +func (s *ClickHouseLogStore) DeleteLogsBatch(ctx context.Context, cutoff time.Time, batchSize int) (int64, error) { + var ids []string + if err := s.db.WithContext(ctx). + Model(&Log{}). + Select("id"). + Where("created_at < ?", cutoff). + Order("created_at ASC"). + Limit(batchSize). + Pluck("id", &ids).Error; err != nil { + return 0, err + } + if len(ids) == 0 { + return 0, nil + } + if err := s.db.WithContext(ctx).Where("id IN ?", ids).Delete(&Log{}).Error; err != nil { + return 0, err + } + return int64(len(ids)), nil +} + +// DeleteExpiredAsyncJobs deletes async jobs whose expiry has passed. +// Overridden for the same reason as DeleteLogsBatch: mutation deletes report +// 0 rows affected, so ids are selected first and their count returned. +func (s *ClickHouseLogStore) DeleteExpiredAsyncJobs(ctx context.Context) (int64, error) { + now := time.Now().UTC() + const batchLimit = 100 + var total int64 + for { + var ids []string + if err := s.db.WithContext(ctx).Model(&AsyncJob{}).Select("id"). + Where("expires_at IS NOT NULL AND expires_at < ?", now). + Limit(batchLimit).Pluck("id", &ids).Error; err != nil { + return total, err + } + if len(ids) == 0 { + return total, nil + } + if err := s.db.WithContext(ctx).Where("id IN ?", ids).Delete(&AsyncJob{}).Error; err != nil { + return total, err + } + total += int64(len(ids)) + if len(ids) < batchLimit { + return total, nil + } + } +} + +// DeleteStaleAsyncJobs deletes processing jobs created before staleSince. +// See DeleteExpiredAsyncJobs for why the count is derived from a prior select. +func (s *ClickHouseLogStore) DeleteStaleAsyncJobs(ctx context.Context, staleSince time.Time) (int64, error) { + const batchLimit = 100 + var total int64 + for { + var ids []string + if err := s.db.WithContext(ctx).Model(&AsyncJob{}).Select("id"). + Where("status = ? AND created_at < ?", "processing", staleSince). + Limit(batchLimit).Pluck("id", &ids).Error; err != nil { + return total, err + } + if len(ids) == 0 { + return total, nil + } + if err := s.db.WithContext(ctx).Where("id IN ?", ids).Delete(&AsyncJob{}).Error; err != nil { + return total, err + } + total += int64(len(ids)) + if len(ids) < batchLimit { + return total, nil + } + } +} + +// UpdateAsyncJob applies a column->value map to an async job row via +// read-modify-write. +func (s *ClickHouseLogStore) UpdateAsyncJob(ctx context.Context, id string, updates map[string]interface{}) error { + st, err := chParseSchema(s.db, &AsyncJob{}) + if err != nil { + return err + } + defer s.lockRMW("async_jobs", id)() + var existing AsyncJob + if err := s.db.WithContext(ctx).Where("id = ?", id).First(&existing).Error; err != nil { + if errors.Is(err, gorm.ErrRecordNotFound) { + return ErrNotFound + } + return err + } + dest := reflect.ValueOf(&existing).Elem() + if err := chApplyUpdateMap(ctx, st, dest, updates); err != nil { + return err + } + return s.chReinsert(ctx, &existing) +} diff --git a/framework/logstore/clickhousestore_test.go b/framework/logstore/clickhousestore_test.go new file mode 100644 index 00000000000..b62099cf02e --- /dev/null +++ b/framework/logstore/clickhousestore_test.go @@ -0,0 +1,727 @@ +package logstore + +import ( + "context" + "fmt" + "reflect" + "strings" + "sync" + "testing" + "time" + + "github.com/maximhq/bifrost/core/schemas" + "github.com/maximhq/bifrost/framework/objectstore" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "gorm.io/gorm" + gormschema "gorm.io/gorm/schema" +) + +// ClickHouse test connection matches the clickhouse service in +// framework/docker-compose.yml (native protocol on host port 9001; host 9000 +// is taken by Weaviate). +const ( + clickhouseTestHost = "localhost" + clickhouseTestPort = "9001" + clickhouseTestDatabase = "bifrost" + clickhouseTestUser = "bifrost" + clickhouseTestPassword = "bifrost_password" +) + +func clickhouseTestConfig() *ClickHouseConfig { + return &ClickHouseConfig{ + Host: schemas.NewSecretVar(clickhouseTestHost), + Port: schemas.NewSecretVar(clickhouseTestPort), + Database: schemas.NewSecretVar(clickhouseTestDatabase), + Username: schemas.NewSecretVar(clickhouseTestUser), + Password: schemas.NewSecretVar(clickhouseTestPassword), + } +} + +// trySetupClickHouseStore connects to the docker-compose ClickHouse, runs +// migrations, and truncates the log tables for a clean slate. Skips the test +// when ClickHouse is unavailable. +func trySetupClickHouseStore(t *testing.T) *ClickHouseLogStore { + t.Helper() + ctx := context.Background() + store, err := newClickHouseLogStore(ctx, clickhouseTestConfig(), 0, testLogger{}) + if err != nil { + t.Skipf("ClickHouse not available, skipping test: %v", err) + } + ch := store.(*ClickHouseLogStore) + for _, table := range []string{"logs", "mcp_tool_logs", "async_jobs"} { + require.NoError(t, ch.db.Exec("TRUNCATE TABLE "+table).Error) + } + t.Cleanup(func() { _ = ch.Close(context.Background()) }) + return ch +} + +func chTestLog(id string, ts time.Time) *Log { + return &Log{ + ID: id, + Timestamp: ts, + Object: "chat.completion", + Provider: "openai", + Model: "gpt-4o", + Status: "processing", + CreatedAt: ts, + } +} + +// chCountRows counts logical rows visible for an id; with the connection-level +// final=1 setting, ReplacingMergeTree duplicates must collapse to one. +func chCountRows(t *testing.T, db *gorm.DB, table, id string) int64 { + t.Helper() + var count int64 + require.NoError(t, db.Raw(fmt.Sprintf("SELECT count() FROM `%s` WHERE id = ?", table), id).Scan(&count).Error) + return count +} + +// --- Pure unit tests (no server required) --- + +func TestBuildClickHouseDSN(t *testing.T) { + t.Run("NativeDefaults", func(t *testing.T) { + dsn, err := buildClickHouseDSN(&ClickHouseConfig{Host: schemas.NewSecretVar("ch.local")}) + require.NoError(t, err) + assert.True(t, strings.HasPrefix(dsn, "clickhouse://ch.local:9000/default?"), dsn) + assert.Contains(t, dsn, "final=1") + assert.Contains(t, dsn, "mutations_sync=1") + assert.Contains(t, dsn, "prefer_column_name_to_alias=1") + assert.Contains(t, dsn, "dial_timeout=10s") + assert.NotContains(t, dsn, "secure=") + }) + + t.Run("NativeSecureUsesTLSPort", func(t *testing.T) { + dsn, err := buildClickHouseDSN(&ClickHouseConfig{Host: schemas.NewSecretVar("ch.local"), Secure: true}) + require.NoError(t, err) + assert.Contains(t, dsn, "ch.local:9440") + assert.Contains(t, dsn, "secure=true") + }) + + t.Run("HTTPProtocol", func(t *testing.T) { + dsn, err := buildClickHouseDSN(&ClickHouseConfig{Host: schemas.NewSecretVar("ch.local"), Protocol: "http"}) + require.NoError(t, err) + assert.True(t, strings.HasPrefix(dsn, "http://ch.local:8123/default?"), dsn) + }) + + t.Run("HTTPSecureUsesHTTPSScheme", func(t *testing.T) { + dsn, err := buildClickHouseDSN(&ClickHouseConfig{Host: schemas.NewSecretVar("ch.local"), Protocol: "http", Secure: true}) + require.NoError(t, err) + assert.True(t, strings.HasPrefix(dsn, "https://ch.local:8443/default?"), dsn) + assert.Contains(t, dsn, "secure=true") + }) + + t.Run("CredentialsPortAndDatabase", func(t *testing.T) { + dsn, err := buildClickHouseDSN(clickhouseTestConfig()) + require.NoError(t, err) + assert.Contains(t, dsn, "bifrost:bifrost_password@localhost:9001/bifrost") + }) + + t.Run("DialTimeoutMilliseconds", func(t *testing.T) { + dsn, err := buildClickHouseDSN(&ClickHouseConfig{Host: schemas.NewSecretVar("ch.local"), DialTimeout: 2500}) + require.NoError(t, err) + assert.Contains(t, dsn, "dial_timeout=2.5s") + }) + + t.Run("MissingHost", func(t *testing.T) { + _, err := buildClickHouseDSN(&ClickHouseConfig{}) + require.Error(t, err) + }) + + t.Run("UnsupportedProtocol", func(t *testing.T) { + _, err := buildClickHouseDSN(&ClickHouseConfig{Host: schemas.NewSecretVar("ch.local"), Protocol: "grpc"}) + require.Error(t, err) + }) +} + +func TestChEscapeIdentifier(t *testing.T) { + assert.Equal(t, "prod_cluster", chEscapeIdentifier("prod_cluster")) + assert.Equal(t, "a``b", chEscapeIdentifier("a`b")) + assert.Equal(t, "````", chEscapeIdentifier("``")) +} + +// chUnitSchemaDB returns a gorm.DB usable for schema parsing without a live +// connection (chParseSchema only needs the naming strategy). +func chUnitSchemaDB() *gorm.DB { + return &gorm.DB{Config: &gorm.Config{NamingStrategy: gormschema.NamingStrategy{}}} +} + +func TestChApplyUpdateMapSkipsDedupKeys(t *testing.T) { + ctx := context.Background() + st, err := chParseSchema(chUnitSchemaDB(), &Log{}) + require.NoError(t, err) + + ts := time.Now().UTC().Truncate(time.Millisecond) + row := *chTestLog("log-1", ts) + dest := reflect.ValueOf(&row).Elem() + + err = chApplyUpdateMap(ctx, st, dest, map[string]interface{}{ + "status": "success", + "id": "hijacked", + "timestamp": ts.Add(time.Hour), + "cost": 0.42, + }) + require.NoError(t, err) + + assert.Equal(t, "success", row.Status) + require.NotNil(t, row.Cost) + assert.Equal(t, 0.42, *row.Cost) + // Dedup key columns must survive untouched. + assert.Equal(t, "log-1", row.ID) + assert.Equal(t, ts, row.Timestamp) +} + +func TestChApplyStructUpdateSkipsDedupKeys(t *testing.T) { + ctx := context.Background() + st, err := chParseSchema(chUnitSchemaDB(), &Log{}) + require.NoError(t, err) + + ts := time.Now().UTC().Truncate(time.Millisecond) + row := *chTestLog("log-1", ts) + dest := reflect.ValueOf(&row).Elem() + + update := Log{ID: "hijacked", Timestamp: ts.Add(time.Hour), Status: "error", Model: "gpt-4o-mini"} + require.NoError(t, chApplyStructUpdate(ctx, st, dest, reflect.ValueOf(&update).Elem())) + + assert.Equal(t, "error", row.Status) + assert.Equal(t, "gpt-4o-mini", row.Model) + assert.Equal(t, "log-1", row.ID) + assert.Equal(t, ts, row.Timestamp) + // Zero-valued fields in the update struct must not clobber existing values. + assert.Equal(t, "openai", row.Provider) +} + +// --- Integration tests (require docker-compose clickhouse) --- + +func TestClickHouseCreateAndFind(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + ts := time.Now().UTC().Truncate(time.Millisecond) + + require.NoError(t, store.CreateIfNotExists(ctx, chTestLog("ch-create-1", ts))) + + found, err := store.FindByID(ctx, "ch-create-1") + require.NoError(t, err) + assert.Equal(t, "openai", found.Provider) + assert.Equal(t, "gpt-4o", found.Model) + assert.Equal(t, "processing", found.Status) + + present, err := store.IsLogEntryPresent(ctx, "ch-create-1") + require.NoError(t, err) + assert.True(t, present) + + _, err = store.FindByID(ctx, "does-not-exist") + assert.ErrorIs(t, err, ErrNotFound) + + hasLogs, err := store.HasLogs(ctx) + require.NoError(t, err) + assert.True(t, hasLogs) + + require.NoError(t, store.Ping(ctx)) +} + +func TestClickHouseIdempotentCreate(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + ts := time.Now().UTC().Truncate(time.Millisecond) + + entry := chTestLog("ch-idem-1", ts) + require.NoError(t, store.CreateIfNotExists(ctx, entry)) + require.NoError(t, store.CreateIfNotExists(ctx, chTestLog("ch-idem-1", ts))) + + // final=1 must collapse the duplicate inserts into a single logical row. + assert.Equal(t, int64(1), chCountRows(t, store.db, "logs", "ch-idem-1")) +} + +func TestClickHouseCreateIfNotExistsKeepsExistingRow(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + ts := time.Now().UTC().Truncate(time.Millisecond) + + require.NoError(t, store.CreateIfNotExists(ctx, chTestLog("ch-keep-1", ts))) + require.NoError(t, store.Update(ctx, "ch-keep-1", map[string]interface{}{ + "status": "success", + "has_object": true, + })) + + // A retried insert of the initial "processing" entry must be a no-op: + // ReplacingMergeTree alone would keep the newest ver and resurrect the + // stale row, dropping status and has_object. + require.NoError(t, store.CreateIfNotExists(ctx, chTestLog("ch-keep-1", ts))) + found, err := store.FindByID(ctx, "ch-keep-1") + require.NoError(t, err) + assert.Equal(t, "success", found.Status) + assert.True(t, found.HasObject) + + // Batch variant: existing id skipped, new id inserted. + require.NoError(t, store.BatchCreateIfNotExists(ctx, []*Log{ + chTestLog("ch-keep-1", ts), + chTestLog("ch-keep-2", ts.Add(time.Millisecond)), + })) + found, err = store.FindByID(ctx, "ch-keep-1") + require.NoError(t, err) + assert.Equal(t, "success", found.Status) + assert.True(t, found.HasObject) + _, err = store.FindByID(ctx, "ch-keep-2") + require.NoError(t, err) + + // MCP variant. + require.NoError(t, store.BatchCreateMCPToolLogsIfNotExists(ctx, []*MCPToolLog{chTestMCPToolLog("ch-keep-mcp-1", ts)})) + require.NoError(t, store.UpdateMCPToolLog(ctx, "ch-keep-mcp-1", map[string]interface{}{"status": "success", "has_object": true})) + require.NoError(t, store.BatchCreateMCPToolLogsIfNotExists(ctx, []*MCPToolLog{chTestMCPToolLog("ch-keep-mcp-1", ts)})) + foundMCP, err := store.FindMCPToolLog(ctx, "ch-keep-mcp-1") + require.NoError(t, err) + assert.Equal(t, "success", foundMCP.Status) + assert.True(t, foundMCP.HasObject) +} + +func TestClickHouseBatchCreate(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + ts := time.Now().UTC().Truncate(time.Millisecond) + + entries := []*Log{ + chTestLog("ch-batch-1", ts), + chTestLog("ch-batch-2", ts.Add(time.Millisecond)), + chTestLog("ch-batch-3", ts.Add(2*time.Millisecond)), + } + require.NoError(t, store.BatchCreateIfNotExists(ctx, entries)) + require.NoError(t, store.BatchCreateIfNotExists(ctx, nil)) // no-op + + for _, id := range []string{"ch-batch-1", "ch-batch-2", "ch-batch-3"} { + _, err := store.FindByID(ctx, id) + require.NoError(t, err) + } +} + +func TestClickHouseUpdateWithMap(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + ts := time.Now().UTC().Truncate(time.Millisecond) + + require.NoError(t, store.CreateIfNotExists(ctx, chTestLog("ch-upd-map", ts))) + require.NoError(t, store.Update(ctx, "ch-upd-map", map[string]interface{}{ + "status": "success", + "cost": 1.25, + })) + + found, err := store.FindByID(ctx, "ch-upd-map") + require.NoError(t, err) + assert.Equal(t, "success", found.Status) + require.NotNil(t, found.Cost) + assert.Equal(t, 1.25, *found.Cost) + // Untouched columns must survive the re-insert. + assert.Equal(t, "openai", found.Provider) + assert.Equal(t, "gpt-4o", found.Model) + assert.Equal(t, int64(1), chCountRows(t, store.db, "logs", "ch-upd-map")) + + assert.ErrorIs(t, store.Update(ctx, "missing-id", map[string]interface{}{"status": "success"}), ErrNotFound) +} + +func TestClickHouseUpdateWithStruct(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + ts := time.Now().UTC().Truncate(time.Millisecond) + + require.NoError(t, store.CreateIfNotExists(ctx, chTestLog("ch-upd-struct", ts))) + + latency := 123.5 + require.NoError(t, store.Update(ctx, "ch-upd-struct", &Log{Status: "success", Latency: &latency})) + + found, err := store.FindByID(ctx, "ch-upd-struct") + require.NoError(t, err) + assert.Equal(t, "success", found.Status) + require.NotNil(t, found.Latency) + assert.Equal(t, 123.5, *found.Latency) + assert.Equal(t, "gpt-4o", found.Model) +} + +func TestClickHouseUpdateCannotRewriteDedupKey(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + ts := time.Now().UTC().Truncate(time.Millisecond) + + require.NoError(t, store.CreateIfNotExists(ctx, chTestLog("ch-upd-key", ts))) + + // An update that tries to move the dedup key must not fork a second + // logical row (the table ORDER BY is (timestamp, id)). + require.NoError(t, store.Update(ctx, "ch-upd-key", map[string]interface{}{ + "timestamp": ts.Add(time.Hour), + "id": "ch-upd-key-forged", + "status": "success", + })) + + assert.Equal(t, int64(1), chCountRows(t, store.db, "logs", "ch-upd-key")) + assert.Equal(t, int64(0), chCountRows(t, store.db, "logs", "ch-upd-key-forged")) + + found, err := store.FindByID(ctx, "ch-upd-key") + require.NoError(t, err) + assert.Equal(t, "success", found.Status) + assert.Equal(t, ts.UnixMilli(), found.Timestamp.UnixMilli()) +} + +func TestClickHouseConcurrentUpdatesPreserveBothPatches(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + + // The object-offload path (has_object) racing the completion path + // (status/cost) is the exact lost-update scenario the per-id RMW locks + // exist for; without them one patch silently vanishes. + for i := 0; i < 10; i++ { + id := fmt.Sprintf("ch-race-%d", i) + ts := time.Now().UTC().Truncate(time.Millisecond) + require.NoError(t, store.CreateIfNotExists(ctx, chTestLog(id, ts))) + + var wg sync.WaitGroup + errs := make([]error, 2) + wg.Add(2) + go func() { + defer wg.Done() + errs[0] = store.Update(ctx, id, map[string]interface{}{"status": "success", "cost": 0.5}) + }() + go func() { + defer wg.Done() + errs[1] = store.Update(ctx, id, map[string]interface{}{"has_object": true}) + }() + wg.Wait() + require.NoError(t, errs[0]) + require.NoError(t, errs[1]) + + found, err := store.FindByID(ctx, id) + require.NoError(t, err) + assert.Equal(t, "success", found.Status, "status patch lost for %s", id) + assert.True(t, found.HasObject, "has_object patch lost for %s", id) + } +} + +func TestClickHouseBulkUpdateCost(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + ts := time.Now().UTC().Truncate(time.Millisecond) + + updates := map[string]float64{} + for i := 0; i < 5; i++ { + id := fmt.Sprintf("ch-cost-%d", i) + require.NoError(t, store.CreateIfNotExists(ctx, chTestLog(id, ts.Add(time.Duration(i)*time.Millisecond)))) + updates[id] = float64(i) * 0.1 + } + // Unknown ids must be ignored, not error. + updates["ch-cost-missing"] = 9.9 + + require.NoError(t, store.BulkUpdateCost(ctx, updates)) + require.NoError(t, store.BulkUpdateCost(ctx, nil)) // no-op + + for i := 0; i < 5; i++ { + id := fmt.Sprintf("ch-cost-%d", i) + found, err := store.FindByID(ctx, id) + require.NoError(t, err) + require.NotNil(t, found.Cost, "cost missing for %s", id) + assert.InDelta(t, float64(i)*0.1, *found.Cost, 1e-9) + assert.Equal(t, int64(1), chCountRows(t, store.db, "logs", id)) + } +} + +func TestClickHouseSearchAndStats(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + ts := time.Now().UTC().Truncate(time.Millisecond) + + for i := 0; i < 3; i++ { + entry := chTestLog(fmt.Sprintf("ch-search-%d", i), ts.Add(time.Duration(i)*time.Second)) + entry.Status = "success" + if i == 2 { + entry.Provider = "anthropic" + entry.Model = "claude-sonnet-4-5" + } + require.NoError(t, store.CreateIfNotExists(ctx, entry)) + } + + result, err := store.SearchLogs(ctx, SearchFilters{}, PaginationOptions{Limit: 10}) + require.NoError(t, err) + assert.Len(t, result.Logs, 3) + + filtered, err := store.SearchLogs(ctx, SearchFilters{Providers: []string{"anthropic"}}, PaginationOptions{Limit: 10}) + require.NoError(t, err) + assert.Len(t, filtered.Logs, 1) + + stats, err := store.GetStats(ctx, SearchFilters{}) + require.NoError(t, err) + assert.Equal(t, int64(3), stats.TotalRequests) + + models, err := store.GetDistinctModels(ctx, 10, "") + require.NoError(t, err) + assert.ElementsMatch(t, []string{"gpt-4o", "claude-sonnet-4-5"}, models) +} + +func TestClickHouseDeleteLogs(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + ts := time.Now().UTC().Truncate(time.Millisecond) + + for _, id := range []string{"ch-del-1", "ch-del-2", "ch-del-3"} { + require.NoError(t, store.CreateIfNotExists(ctx, chTestLog(id, ts))) + } + + require.NoError(t, store.DeleteLog(ctx, "ch-del-1")) + require.NoError(t, store.DeleteLogs(ctx, []string{"ch-del-2", "ch-del-3"})) + + for _, id := range []string{"ch-del-1", "ch-del-2", "ch-del-3"} { + _, err := store.FindByID(ctx, id) + assert.ErrorIs(t, err, ErrNotFound, "log %s should be deleted", id) + } +} + +func TestClickHouseDeleteLogsBatch(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + + old := time.Now().UTC().Add(-48 * time.Hour).Truncate(time.Millisecond) + fresh := time.Now().UTC().Truncate(time.Millisecond) + require.NoError(t, store.CreateIfNotExists(ctx, chTestLog("ch-old", old))) + require.NoError(t, store.CreateIfNotExists(ctx, chTestLog("ch-fresh", fresh))) + + deleted, err := store.DeleteLogsBatch(ctx, time.Now().UTC().Add(-24*time.Hour), 100) + require.NoError(t, err) + assert.Equal(t, int64(1), deleted, "cleaner pacing relies on an accurate deleted count") + + _, err = store.FindByID(ctx, "ch-old") + assert.ErrorIs(t, err, ErrNotFound) + _, err = store.FindByID(ctx, "ch-fresh") + assert.NoError(t, err) +} + +func chTestMCPToolLog(id string, ts time.Time) *MCPToolLog { + return &MCPToolLog{ + ID: id, + Timestamp: ts, + ToolName: "search_web", + Status: "processing", + CreatedAt: ts, + } +} + +func TestClickHouseMCPToolLogs(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + ts := time.Now().UTC().Truncate(time.Millisecond) + + entries := []*MCPToolLog{ + chTestMCPToolLog("ch-mcp-1", ts), + chTestMCPToolLog("ch-mcp-2", ts.Add(time.Millisecond)), + } + require.NoError(t, store.BatchCreateMCPToolLogsIfNotExists(ctx, entries)) + require.NoError(t, store.BatchCreateMCPToolLogsIfNotExists(ctx, nil)) // no-op + + found, err := store.FindMCPToolLog(ctx, "ch-mcp-1") + require.NoError(t, err) + assert.Equal(t, "search_web", found.ToolName) + + // Map update. + latency := 42.0 + require.NoError(t, store.UpdateMCPToolLog(ctx, "ch-mcp-1", map[string]interface{}{ + "status": "success", + "latency": latency, + })) + found, err = store.FindMCPToolLog(ctx, "ch-mcp-1") + require.NoError(t, err) + assert.Equal(t, "success", found.Status) + require.NotNil(t, found.Latency) + assert.Equal(t, 42.0, *found.Latency) + assert.Equal(t, int64(1), chCountRows(t, store.db, "mcp_tool_logs", "ch-mcp-1")) + + // Struct update preserves untouched fields and the dedup key. + require.NoError(t, store.UpdateMCPToolLog(ctx, "ch-mcp-2", &MCPToolLog{Status: "error", Timestamp: ts.Add(time.Hour)})) + found, err = store.FindMCPToolLog(ctx, "ch-mcp-2") + require.NoError(t, err) + assert.Equal(t, "error", found.Status) + assert.Equal(t, "search_web", found.ToolName) + assert.Equal(t, ts.Add(time.Millisecond).UnixMilli(), found.Timestamp.UnixMilli()) + assert.Equal(t, int64(1), chCountRows(t, store.db, "mcp_tool_logs", "ch-mcp-2")) + + assert.ErrorIs(t, store.UpdateMCPToolLog(ctx, "missing-id", map[string]interface{}{"status": "success"}), ErrNotFound) + + hasLogs, err := store.HasMCPToolLogs(ctx) + require.NoError(t, err) + assert.True(t, hasLogs) + + result, err := store.SearchMCPToolLogs(ctx, MCPToolLogSearchFilters{}, PaginationOptions{Limit: 10}) + require.NoError(t, err) + assert.Len(t, result.Logs, 2) +} + +// TestClickHouseHybridHasObjectSurvivesDuplicateCreate exercises the full +// HybridLogStore-over-ClickHouse flow that hybrid mode depends on: create a +// payload-bearing entry, let the async upload worker flip has_object, apply +// the completion update, then retry the initial create. The completed status, +// the has_object flag, and payload hydration must all survive the retry. +func TestClickHouseHybridHasObjectSurvivesDuplicateCreate(t *testing.T) { + ch := trySetupClickHouseStore(t) + objStore := objectstore.NewInMemoryObjectStore() + hybrid := newHybridLogStore(ch, objStore, "test", hybridTestLogger{}, nil) + ctx := context.Background() + ts := time.Now().UTC().Truncate(time.Millisecond) + + input := "hello from clickhouse hybrid" + mkEntry := func() *Log { + entry := chTestLog("ch-hybrid-1", ts) + entry.InputHistoryParsed = []schemas.ChatMessage{ + {Role: schemas.ChatMessageRoleUser, Content: &schemas.ChatMessageContent{ContentStr: &input}}, + } + return entry + } + + require.NoError(t, hybrid.CreateIfNotExists(ctx, mkEntry())) + + // The upload worker sets has_object asynchronously after the S3 put. + waitForUploads(t, func() bool { + log, err := ch.FindByID(ctx, "ch-hybrid-1") + return err == nil && log.HasObject + }) + + // Completion update from the logging plugin's write path. + require.NoError(t, hybrid.Update(ctx, "ch-hybrid-1", map[string]interface{}{"status": "success"})) + + // Duplicate create retry must not resurrect the stale processing row. + require.NoError(t, hybrid.CreateIfNotExists(ctx, mkEntry())) + + found, err := hybrid.FindByID(ctx, "ch-hybrid-1") + require.NoError(t, err) + assert.Equal(t, "success", found.Status) + assert.True(t, found.HasObject) + assert.NotEmpty(t, found.InputHistory, "payload should hydrate from the object store") + assert.Contains(t, found.ContentSummary, input) + + require.NoError(t, hybrid.Close(ctx)) +} + +// TestClickHouseNodeUsageCursorDoesNotRewind guards the budget-usage gossip +// cursor against the GORM ClickHouse driver's seconds-truncation of time.Time +// args: a truncated cursor bound would rewind to the start of its second and +// double-count every row already aggregated on the previous scan. +func TestClickHouseNodeUsageCursorDoesNotRewind(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + // Sub-second offsets are the point of this test: rows land mid-second. + base := time.Now().UTC().Truncate(time.Second).Add(288 * time.Millisecond) + + nodeID := "node-1" + budgetIDs := `["b1"]` + mk := func(id string, ts time.Time, cost float64) *Log { + l := chTestLog(id, ts) + l.Status = "success" + l.ClusterNodeID = &nodeID + l.BudgetIDs = &budgetIDs + l.Cost = &cost + return l + } + require.NoError(t, store.CreateIfNotExists(ctx, mk("ch-usage-1", base, 1.0))) + require.NoError(t, store.CreateIfNotExists(ctx, mk("ch-usage-2", base.Add(200*time.Millisecond), 2.0))) + + first, err := store.GetNodeUsageAfter(ctx, nodeID, NodeUsageCursor{Timestamp: base.Add(-time.Hour)}) + require.NoError(t, err) + assert.Equal(t, 2, first.RowCount) + assert.InDelta(t, 3.0, first.BudgetCosts["b1"], 1e-9) + + // Re-scan from the advanced cursor: nothing new, so nothing may be + // re-aggregated - a rewound (seconds-truncated) cursor would return both + // rows again and double-count the budget spend. + second, err := store.GetNodeUsageAfter(ctx, nodeID, first.NextCursor) + require.NoError(t, err) + assert.Equal(t, 0, second.RowCount, "cursor must not rewind into already-counted rows") + assert.Empty(t, second.BudgetCosts) +} + +func TestClickHouseAsyncJobs(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + now := time.Now().UTC().Truncate(time.Millisecond) + + job := &AsyncJob{ + ID: "ch-job-1", + Status: schemas.AsyncJobStatusProcessing, + RequestType: schemas.ChatCompletionRequest, + CreatedAt: now, + } + require.NoError(t, store.CreateAsyncJob(ctx, job)) + + found, err := store.FindAsyncJobByID(ctx, "ch-job-1") + require.NoError(t, err) + assert.Equal(t, schemas.AsyncJobStatusProcessing, found.Status) + + completedAt := now.Add(time.Second) + require.NoError(t, store.UpdateAsyncJob(ctx, "ch-job-1", map[string]interface{}{ + "status": string(schemas.AsyncJobStatusCompleted), + "response": `{"ok":true}`, + "completed_at": completedAt, + })) + found, err = store.FindAsyncJobByID(ctx, "ch-job-1") + require.NoError(t, err) + assert.Equal(t, schemas.AsyncJobStatusCompleted, found.Status) + assert.Equal(t, `{"ok":true}`, found.Response) + assert.Equal(t, int64(1), chCountRows(t, store.db, "async_jobs", "ch-job-1")) + + // Expired job cleanup. + expiredAt := now.Add(-time.Hour) + expired := &AsyncJob{ + ID: "ch-job-expired", + Status: schemas.AsyncJobStatusCompleted, + RequestType: schemas.ChatCompletionRequest, + ExpiresAt: &expiredAt, + CreatedAt: now.Add(-2 * time.Hour), + } + require.NoError(t, store.CreateAsyncJob(ctx, expired)) + _, err = store.DeleteExpiredAsyncJobs(ctx) + require.NoError(t, err) + _, err = store.FindAsyncJobByID(ctx, "ch-job-expired") + assert.Error(t, err, "expired job should be deleted") + + // Stale processing job cleanup. + stale := &AsyncJob{ + ID: "ch-job-stale", + Status: schemas.AsyncJobStatusProcessing, + RequestType: schemas.ChatCompletionRequest, + CreatedAt: now.Add(-48 * time.Hour), + } + require.NoError(t, store.CreateAsyncJob(ctx, stale)) + _, err = store.DeleteStaleAsyncJobs(ctx, now.Add(-24*time.Hour)) + require.NoError(t, err) + _, err = store.FindAsyncJobByID(ctx, "ch-job-stale") + assert.Error(t, err, "stale processing job should be deleted") + + // The completed job must survive both cleanups. + _, err = store.FindAsyncJobByID(ctx, "ch-job-1") + assert.NoError(t, err) +} + +func TestClickHouseHistograms(t *testing.T) { + store := trySetupClickHouseStore(t) + ctx := context.Background() + ts := time.Now().UTC().Truncate(time.Millisecond) + + for i := 0; i < 4; i++ { + entry := chTestLog(fmt.Sprintf("ch-hist-%d", i), ts.Add(time.Duration(i)*time.Second)) + entry.Status = "success" + cost := 0.25 + entry.Cost = &cost + entry.TotalTokens = 100 + entry.PromptTokens = 60 + entry.CompletionTokens = 40 + require.NoError(t, store.CreateIfNotExists(ctx, entry)) + } + + hist, err := store.GetHistogram(ctx, SearchFilters{}, 60) + require.NoError(t, err) + require.NotNil(t, hist) + + costHist, err := store.GetCostHistogram(ctx, SearchFilters{}, 60) + require.NoError(t, err) + require.NotNil(t, costHist) + + tokenHist, err := store.GetTokenHistogram(ctx, SearchFilters{}, 60) + require.NoError(t, err) + require.NotNil(t, tokenHist) + + modelRankings, err := store.GetModelRankings(ctx, SearchFilters{}) + require.NoError(t, err) + require.NotNil(t, modelRankings) +} diff --git a/framework/logstore/config.go b/framework/logstore/config.go index b784aa59772..458c046ff00 100644 --- a/framework/logstore/config.go +++ b/framework/logstore/config.go @@ -110,6 +110,12 @@ func (c *Config) UnmarshalJSON(data []byte) error { return fmt.Errorf("failed to unmarshal postgres config: %w", err) } c.Config = &postgresConfig + case LogStoreTypeClickHouse: + var clickhouseConfig ClickHouseConfig + if err := json.Unmarshal(temp.Config, &clickhouseConfig); err != nil { + return fmt.Errorf("failed to unmarshal clickhouse config: %w", err) + } + c.Config = &clickhouseConfig default: return fmt.Errorf("unknown log store type: %s", temp.Type) } diff --git a/framework/logstore/dialectsql.go b/framework/logstore/dialectsql.go new file mode 100644 index 00000000000..e8b610cfe20 --- /dev/null +++ b/framework/logstore/dialectsql.go @@ -0,0 +1,27 @@ +package logstore + +import "fmt" + +// unixBucketExpr returns a SQL expression that truncates the `timestamp` column +// to a bucket boundary and yields an integer unix-seconds value, per dialect. +// The returned string is a complete SQL fragment (the sqlite branch contains a +// literal strftime '%s' specifier, which is safe to pass as an argument to a +// later fmt.Sprintf since only the format string is scanned for verbs). +// +// Keeping the per-dialect bucket math in one place lets the ~17 histogram +// queries in rdb.go share a single, dialect-correct expression instead of +// branching inline (which previously routed ClickHouse into the Postgres +// EXTRACT(EPOCH ...) path - invalid ClickHouse SQL). +func unixBucketExpr(dialect string, bucketSizeSeconds int64) string { + switch dialect { + case "sqlite": + return fmt.Sprintf("(CAST(strftime('%%s', timestamp) AS INTEGER) / %d) * %d", bucketSizeSeconds, bucketSizeSeconds) + case "mysql": + return fmt.Sprintf("(FLOOR(UNIX_TIMESTAMP(timestamp) / %d) * %d)", bucketSizeSeconds, bucketSizeSeconds) + case "clickhouse": + return fmt.Sprintf("toInt64(intDiv(toUnixTimestamp(timestamp), %d) * %d)", bucketSizeSeconds, bucketSizeSeconds) + default: + // PostgreSQL (and others) + return fmt.Sprintf("CAST(FLOOR(EXTRACT(EPOCH FROM timestamp) / %d) * %d AS BIGINT)", bucketSizeSeconds, bucketSizeSeconds) + } +} diff --git a/framework/logstore/logstoreparity_test.go b/framework/logstore/logstoreparity_test.go new file mode 100644 index 00000000000..b0cbe7046f2 --- /dev/null +++ b/framework/logstore/logstoreparity_test.go @@ -0,0 +1,1095 @@ +package logstore + +// Three-way LogStore parity suite: seeds identical fixtures into SQLite, +// Postgres (raw path, no matviews), and ClickHouse, runs every LogStore +// interface method on all three, and asserts that the non-reference backends +// return results equal to Postgres (the reference). This is the executable +// form of "every backend is 100% compatible": all 60 interface methods are +// exercised - reads, writes, mutations, async jobs, and deletes - including +// every dialect branch in rdb.go (metadata JSON, cache-hit extraction, +// histogram bucket math, routing-engine matching, distinct queries) and the +// DAC queryscope path. +// +// The Postgres store is built bare (no ensureMatViews), so matViewsReady stays +// false and Postgres deterministically takes the same raw-table path the +// matviews approximate. Known, deliberate divergences are NOT asserted here: +// multi-value team_ids array matching and team/BU dimension fan-out +// (postgres-only features; fixtures carry scalar ids only - the fan-out's +// attributed-totals metadata is excluded from the contract), ILIKE +// case-insensitivity (fixtures use exact case), FTS-vs-LIKE content search +// semantics (the fixture term matches under all three), and inc_number +// (Postgres-assigned; NULL elsewhere - excluded from projections). +// +// Postgres (5432) and ClickHouse (native 9001) must be reachable via +// framework/docker-compose.yml; the suite skips otherwise. SQLite uses a temp +// file. Subtests mutate shared state and run in declaration order - do not +// reorder or parallelize them. + +import ( + "context" + "encoding/json" + "fmt" + "math" + "path/filepath" + "sort" + "testing" + "time" + + "github.com/maximhq/bifrost/core/schemas" + "github.com/maximhq/bifrost/framework/queryscope" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "gorm.io/gorm" +) + +// parityReference is the backend the others are compared against. +const parityReference = "postgres" + +// parityBackends connects all three backends (skipping when Postgres or +// ClickHouse is unavailable), resets their tables, and returns them keyed by +// name. +func parityBackends(t *testing.T) map[string]LogStore { + t.Helper() + ch := trySetupClickHouseStore(t) + + db := trySetupPostgresDB(t) + if db == nil { + t.Skip("Postgres not available, skipping parity test") + } + // Clean slate - same approach as setupPerfTestDB. Matviews are dropped and + // never recreated, so matViewsReady stays false and every Postgres query + // takes the raw-table path (the honest comparison target). + dropAllManagedMatViews(db) + require.NoError(t, db.Exec("DROP TABLE IF EXISTS mcp_tool_logs CASCADE").Error) + require.NoError(t, db.Exec("DROP TABLE IF EXISTS async_jobs CASCADE").Error) + require.NoError(t, db.Exec("DROP TABLE IF EXISTS logs CASCADE").Error) + require.NoError(t, db.Exec("CREATE TABLE IF NOT EXISTS migrations (id VARCHAR(255) PRIMARY KEY)").Error) + require.NoError(t, db.Exec("DELETE FROM migrations").Error) + require.NoError(t, triggerMigrations(context.Background(), db, testLogger{})) + pg := &RDBLogStore{db: db, logger: testLogger{}} + + sq, err := newSqliteLogStore(context.Background(), &SQLiteConfig{ + Path: filepath.Join(t.TempDir(), "parity.db"), + }, testLogger{}) + require.NoError(t, err) + + return map[string]LogStore{"sqlite": sq, parityReference: pg, "clickhouse": ch} +} + +func strPtrP(s string) *string { return &s } +func f64PtrP(f float64) *float64 { return &f } +func intPtrP(i int) *int { return &i } +func timePtrP(ts time.Time) *time.Time { return &ts } + +// parityLogSpec builds one fixture row. All timestamps are whole seconds so +// the GORM ClickHouse driver's seconds-precision formatting of time.Time +// filter args cannot skew range comparisons. +type parityLogSpec struct { + id string + offsetSec int // subtracted from base + object string + provider string + model string + status string + alias *string + selectedKey string + vkID, vkName *string + teamID *string + customerID *string + buID *string + userID *string + cost *float64 + latency *float64 + tokens [3]int // prompt, completion, total + stopReason *string + routing *string + metadata *string + cacheDebug string + content string + parentID *string + nodeID *string + budgetIDs *string + rateLimitIDs *string +} + +func (s parityLogSpec) toLog(base time.Time) *Log { + ts := base.Add(-time.Duration(s.offsetSec) * time.Second) + return &Log{ + ID: s.id, + Timestamp: ts, + Object: s.object, + Provider: s.provider, + Model: s.model, + Status: s.status, + Alias: s.alias, + SelectedKeyID: s.selectedKey, + VirtualKeyID: s.vkID, + VirtualKeyName: s.vkName, + TeamID: s.teamID, + CustomerID: s.customerID, + BusinessUnitID: s.buID, + UserID: s.userID, + Cost: s.cost, + Latency: s.latency, + PromptTokens: s.tokens[0], + CompletionTokens: s.tokens[1], + TotalTokens: s.tokens[2], + StopReason: s.stopReason, + RoutingEnginesUsedStr: s.routing, + Metadata: s.metadata, + CacheDebug: s.cacheDebug, + ContentSummary: s.content, + ParentRequestID: s.parentID, + ClusterNodeID: s.nodeID, + BudgetIDs: s.budgetIDs, + RateLimitIDs: s.rateLimitIDs, + CreatedAt: ts, + } +} + +func paritySpecs() []parityLogSpec { + return []parityLogSpec{ + {id: "p1", offsetSec: 100, object: "chat.completion", provider: "openai", model: "gpt-4o", status: "success", + alias: strPtrP("a1"), selectedKey: "sk1", vkID: strPtrP("vk1"), vkName: strPtrP("VK One"), + teamID: strPtrP("t1"), customerID: strPtrP("c1"), buID: strPtrP("b1"), userID: strPtrP("u1"), + cost: f64PtrP(0.5), latency: f64PtrP(100), tokens: [3]int{100, 50, 150}, stopReason: strPtrP("stop"), + routing: strPtrP("governance,loadbalancing"), metadata: strPtrP(`{"env":"prod"}`), + cacheDebug: `{"hit_type":"direct"}`, content: "alpha bravo hello", parentID: strPtrP("sess1")}, + {id: "p2", offsetSec: 90, object: "chat.completion", provider: "openai", model: "gpt-4o", status: "success", + vkID: strPtrP("vk1"), vkName: strPtrP("VK One"), teamID: strPtrP("t1"), userID: strPtrP("u2"), + cost: f64PtrP(1.25), latency: f64PtrP(250), tokens: [3]int{200, 100, 300}, stopReason: strPtrP("length"), + routing: strPtrP("governance"), metadata: strPtrP(`{"env":"dev"}`), + cacheDebug: `{"hit_type":"semantic"}`, content: "charlie delta", parentID: strPtrP("sess1")}, + {id: "p3", offsetSec: 80, object: "chat.completion", provider: "openai", model: "gpt-4o-mini", status: "error", + vkID: strPtrP("vk2"), vkName: strPtrP("VK Two"), teamID: strPtrP("t2"), userID: strPtrP("u2"), + latency: f64PtrP(50), content: "echo error", parentID: strPtrP("sess1")}, + {id: "p4", offsetSec: 70, object: "chat.completion", provider: "anthropic", model: "claude-3", status: "success", + vkID: strPtrP("vk2"), vkName: strPtrP("VK Two"), teamID: strPtrP("t2"), userID: strPtrP("u3"), + cost: f64PtrP(2.5), latency: f64PtrP(400), tokens: [3]int{400, 100, 500}, stopReason: strPtrP("tool_calls"), + metadata: strPtrP(`{"env":"prod","region":"us"}`), content: "echo foxtrot"}, + {id: "p5", offsetSec: 60, object: "chat.completion", provider: "anthropic", model: "claude-3", status: "processing", + vkID: strPtrP("vk1"), vkName: strPtrP("VK One"), teamID: strPtrP("t1"), userID: strPtrP("u1")}, + {id: "p6", offsetSec: 50, object: "embedding", provider: "openai", model: "gpt-4o", status: "success", + cost: f64PtrP(0), latency: f64PtrP(75), tokens: [3]int{10, 5, 15}}, + {id: "p7", offsetSec: 40, object: "chat.completion", provider: "mistral", model: "mistral-small", status: "success", + alias: strPtrP("a2"), vkID: strPtrP("vk3"), vkName: strPtrP("VK Three"), customerID: strPtrP("c2"), + buID: strPtrP("b2"), userID: strPtrP("u4"), cost: f64PtrP(3.0), latency: f64PtrP(800), + tokens: [3]int{900, 100, 1000}, stopReason: strPtrP("stop"), content: "golf hotel"}, + {id: "p8", offsetSec: 30, object: "chat.completion", provider: "openai", model: "gpt-4o", status: "success", + vkID: strPtrP("vk3"), vkName: strPtrP("VK Three"), teamID: strPtrP("t3"), userID: strPtrP("u1"), + cost: f64PtrP(0.75), latency: f64PtrP(120), tokens: [3]int{50, 25, 75}, stopReason: strPtrP("content_filter"), + routing: strPtrP("routing-rule"), metadata: strPtrP(`{"env":"prod"}`), content: "india juliet"}, + // Cluster-governance rows for GetNodeUsageAfter parity. + {id: "p9", offsetSec: 20, object: "chat.completion", provider: "openai", model: "gpt-4o", status: "success", + cost: f64PtrP(1.0), latency: f64PtrP(90), tokens: [3]int{40, 20, 60}, + nodeID: strPtrP("pnode"), budgetIDs: strPtrP(`["bud1"]`), rateLimitIDs: strPtrP(`["rl1"]`)}, + {id: "p10", offsetSec: 10, object: "chat.completion", provider: "openai", model: "gpt-4o", status: "success", + cost: f64PtrP(2.0), latency: f64PtrP(110), tokens: [3]int{80, 40, 120}, + nodeID: strPtrP("pnode"), budgetIDs: strPtrP(`["bud1","bud2"]`), rateLimitIDs: strPtrP(`["rl1"]`)}, + } +} + +func parityMCPLogs(base time.Time) []*MCPToolLog { + mk := func(id string, offsetSec int, tool, label, status string, latency, cost *float64, vkID, vkName *string) *MCPToolLog { + ts := base.Add(-time.Duration(offsetSec) * time.Second) + return &MCPToolLog{ + ID: id, Timestamp: ts, ToolName: tool, ServerLabel: label, Status: status, + Latency: latency, Cost: cost, VirtualKeyID: vkID, VirtualKeyName: vkName, CreatedAt: ts, + } + } + return []*MCPToolLog{ + mk("m1", 95, "search_web", "srv1", "success", f64PtrP(120), f64PtrP(0.01), strPtrP("vk1"), strPtrP("VK One")), + mk("m2", 85, "search_web", "srv2", "error", f64PtrP(80), nil, strPtrP("vk2"), strPtrP("VK Two")), + mk("m3", 75, "calculator", "srv1", "success", f64PtrP(30), f64PtrP(0.002), strPtrP("vk1"), strPtrP("VK One")), + mk("m4", 65, "calculator", "srv1", "processing", nil, nil, nil, nil), + } +} + +// --- Tolerant deep comparison over JSON-normalized values --- + +// jsonNormalize round-trips v through JSON so every store's results reduce to +// the same generic shape (maps/slices/float64/string/bool/nil). +func jsonNormalize(t *testing.T, v any) any { + t.Helper() + data, err := json.Marshal(v) + require.NoError(t, err) + var out any + require.NoError(t, json.Unmarshal(data, &out)) + return out +} + +// canonicalizeOrder makes presentation-only ordering deterministic before +// diffing: plain string lists (provider/model/dimension series - the buckets +// key their data by name, so list order is cosmetic) are sorted, and +// "rankings" arrays are sorted by a canonical rendering of each entry because +// backends order ties arbitrarily (ORDER BY SUM(...) DESC with equal sums has +// no deterministic tiebreaker on any backend). +func canonicalizeOrder(key string, v any) any { + switch tv := v.(type) { + case map[string]any: + for k, val := range tv { + tv[k] = canonicalizeOrder(k, val) + } + return tv + case []any: + allStrings := len(tv) > 0 + for _, e := range tv { + if _, ok := e.(string); !ok { + allStrings = false + break + } + } + if allStrings { + sort.Slice(tv, func(i, j int) bool { return tv[i].(string) < tv[j].(string) }) + return tv + } + for i := range tv { + tv[i] = canonicalizeOrder("", tv[i]) + } + if key == "rankings" { + keys := make([]string, len(tv)) + for i, e := range tv { + keys[i] = canonicalEntryKey(e) + } + sort.SliceStable(tv, func(i, j int) bool { return keys[i] < keys[j] }) + } + return tv + default: + return v + } +} + +// canonicalEntryKey renders a ranking entry as a stable sort key: sorted map +// keys with floats rounded so sub-tolerance drift cannot reorder entries. +func canonicalEntryKey(v any) string { + m, ok := v.(map[string]any) + if !ok { + return fmt.Sprint(v) + } + keys := make([]string, 0, len(m)) + for k := range m { + keys = append(keys, k) + } + sort.Strings(keys) + out := "" + for _, k := range keys { + switch val := m[k].(type) { + case float64: + out += fmt.Sprintf("%s=%.3f;", k, val) + default: + out += fmt.Sprintf("%s=%v;", k, val) + } + } + return out +} + +// collectDiffs walks two JSON-normalized values and records human-readable +// differences. Numbers compare with a relative/absolute tolerance so +// aggregation-order float drift and percentile interpolation differences +// don't read as incompatibility. +func collectDiffs(path string, a, b any, tol float64, diffs *[]string) { + switch av := a.(type) { + case map[string]any: + bv, ok := b.(map[string]any) + if !ok { + *diffs = append(*diffs, fmt.Sprintf("%s: type mismatch %T vs %T", path, a, b)) + return + } + keys := map[string]struct{}{} + for k := range av { + keys[k] = struct{}{} + } + for k := range bv { + keys[k] = struct{}{} + } + for k := range keys { + aval, aok := av[k] + bval, bok := bv[k] + if aok != bok { + *diffs = append(*diffs, fmt.Sprintf("%s.%s: key presence mismatch", path, k)) + continue + } + collectDiffs(path+"."+k, aval, bval, tol, diffs) + } + case []any: + bv, ok := b.([]any) + if !ok { + *diffs = append(*diffs, fmt.Sprintf("%s: type mismatch %T vs %T", path, a, b)) + return + } + if len(av) != len(bv) { + *diffs = append(*diffs, fmt.Sprintf("%s: length %d vs %d", path, len(av), len(bv))) + return + } + for i := range av { + collectDiffs(fmt.Sprintf("%s[%d]", path, i), av[i], bv[i], tol, diffs) + } + case float64: + bv, ok := b.(float64) + if !ok { + *diffs = append(*diffs, fmt.Sprintf("%s: %v vs %v", path, a, b)) + return + } + limit := math.Max(tol, tol*math.Max(math.Abs(av), math.Abs(bv))) + if math.Abs(av-bv) > limit { + *diffs = append(*diffs, fmt.Sprintf("%s: %v vs %v", path, av, bv)) + } + default: + if !assert.ObjectsAreEqual(a, b) { + *diffs = append(*diffs, fmt.Sprintf("%s: %v vs %v", path, a, b)) + } + } +} + +// assertParity runs f against every backend and requires each non-reference +// backend's result to tolerantly equal the Postgres reference, reporting +// per-field paths on mismatch. +func assertParity(t *testing.T, stores map[string]LogStore, tol float64, f func(context.Context, LogStore) (any, error)) { + t.Helper() + ctx := context.Background() + refVal, err := f(ctx, stores[parityReference]) + require.NoError(t, err, parityReference) + refNorm := canonicalizeOrder("", jsonNormalize(t, refVal)) + for name, s := range stores { + if name == parityReference { + continue + } + val, err := f(ctx, s) + require.NoError(t, err, name) + var diffs []string + collectDiffs("$", refNorm, canonicalizeOrder("", jsonNormalize(t, val)), tol, &diffs) + assert.Empty(t, diffs, "%s vs %s mismatch", parityReference, name) + } +} + +// runOnAll executes op on every backend, failing on any error. Used for the +// write-path phases where the operation itself is the subject. +func runOnAll(t *testing.T, stores map[string]LogStore, op func(context.Context, LogStore) error) { + t.Helper() + for name, s := range stores { + require.NoError(t, op(context.Background(), s), name) + } +} + +// searchProjection reduces a SearchResult to the fields all backends must +// agree on, dropping backend-managed noise (inc_number is Postgres-assigned +// and NULL elsewhere by design). +func searchProjection(r *SearchResult) any { + logs := make([]map[string]any, 0, len(r.Logs)) + for _, l := range r.Logs { + logs = append(logs, logProjection(&l)) + } + return map[string]any{ + "total": r.Pagination.TotalCount, + "has_logs": r.HasLogs, + "stats": r.Stats, + "logs": logs, + } +} + +func logProjection(l *Log) map[string]any { + return map[string]any{ + "id": l.ID, "ts_ms": l.Timestamp.UnixMilli(), "object": l.Object, + "provider": l.Provider, "model": l.Model, "status": l.Status, + "alias": l.Alias, "selected_key_id": l.SelectedKeyID, + "virtual_key_id": l.VirtualKeyID, "team_id": l.TeamID, + "customer_id": l.CustomerID, "business_unit_id": l.BusinessUnitID, + "user_id": l.UserID, "cost": l.Cost, "latency": l.Latency, + "prompt_tokens": l.PromptTokens, "completion_tokens": l.CompletionTokens, + "total_tokens": l.TotalTokens, "stop_reason": l.StopReason, + "content_summary": l.ContentSummary, + } +} + +func mcpProjection(logs []MCPToolLog) []map[string]any { + out := make([]map[string]any, 0, len(logs)) + for _, l := range logs { + out = append(out, map[string]any{ + "id": l.ID, "ts_ms": l.Timestamp.UnixMilli(), "tool": l.ToolName, + "label": l.ServerLabel, "status": l.Status, "latency": l.Latency, + "cost": l.Cost, "virtual_key_id": l.VirtualKeyID, + }) + } + return out +} + +func asyncJobProjection(j *AsyncJob) map[string]any { + out := map[string]any{ + "id": j.ID, "status": j.Status, "request_type": j.RequestType, + "response": j.Response, "status_code": j.StatusCode, "error": j.Error, + "virtual_key_id": j.VirtualKeyID, "created_ms": j.CreatedAt.UnixMilli(), + } + if j.ExpiresAt != nil { + out["expires_ms"] = j.ExpiresAt.UnixMilli() + } + return out +} + +// remainingLogIDs lists surviving log ids - the post-state assertion used +// after every destructive operation. +func remainingLogIDs(ctx context.Context, s LogStore) (any, error) { + logs, err := s.FindAll(ctx, map[string]interface{}{}, "id") + if err != nil { + return nil, err + } + ids := make([]string, 0, len(logs)) + for _, l := range logs { + ids = append(ids, l.ID) + } + sort.Strings(ids) + return ids, nil +} + +func remainingMCPIDs(ctx context.Context, s LogStore) (any, error) { + r, err := s.SearchMCPToolLogs(ctx, MCPToolLogSearchFilters{}, PaginationOptions{Limit: 100, SortBy: "timestamp", Order: "desc"}) + if err != nil { + return nil, err + } + ids := make([]string, 0, len(r.Logs)) + for _, l := range r.Logs { + ids = append(ids, l.ID) + } + sort.Strings(ids) + return ids, nil +} + +func TestLogStoreParity(t *testing.T) { + stores := parityBackends(t) + ctx := context.Background() + base := time.Now().UTC().Truncate(time.Second) + + // --- Phase: create paths (Create, CreateIfNotExists, batch variants) --- + + specs := paritySpecs() + runOnAll(t, stores, func(ctx context.Context, s LogStore) error { + // p1 through the singular paths, the rest through the batch path, so + // Create, CreateIfNotExists, and BatchCreateIfNotExists all execute. + if err := s.Create(ctx, specs[0].toLog(base)); err != nil { + return err + } + if err := s.CreateIfNotExists(ctx, specs[1].toLog(base)); err != nil { + return err + } + var rest []*Log + for _, spec := range specs[2:] { + rest = append(rest, spec.toLog(base)) + } + if err := s.BatchCreateIfNotExists(ctx, rest); err != nil { + return err + } + mcp := parityMCPLogs(base) + if err := s.CreateMCPToolLog(ctx, mcp[0]); err != nil { + return err + } + return s.BatchCreateMCPToolLogsIfNotExists(ctx, mcp[1:]) + }) + + t.Run("Ping", func(t *testing.T) { + runOnAll(t, stores, func(ctx context.Context, s LogStore) error { return s.Ping(ctx) }) + }) + + windowStart := base.Add(-2 * time.Minute) + windowEnd := base.Add(time.Minute) + window := SearchFilters{StartTime: timePtrP(windowStart), EndTime: timePtrP(windowEnd)} + page := PaginationOptions{Limit: 50, SortBy: "timestamp", Order: "desc"} + + // --- Phase: reads --- + + searchCases := map[string]SearchFilters{ + "all": window, + "providers": {Providers: []string{"openai"}}, + "models": {Models: []string{"gpt-4o", "claude-3"}}, + "status": {Status: []string{"success"}}, + "stop_reasons": {StopReasons: []string{"stop"}}, + "objects": {Objects: []string{"embedding"}}, + "aliases": {Aliases: []string{"a1"}}, + "selected_keys": {SelectedKeyIDs: []string{"sk1"}}, + "virtual_keys": {VirtualKeyIDs: []string{"vk1"}}, + "teams": {TeamIDs: []string{"t1", "t3"}}, + "customers": {CustomerIDs: []string{"c1"}}, + "users": {UserIDs: []string{"u1"}}, + "business_units": {BusinessUnitIDs: []string{"b1"}}, + "routing_engines": {RoutingEngineUsed: []string{"loadbalancing", "routing-rule"}}, + "time_range": {StartTime: timePtrP(base.Add(-75 * time.Second)), EndTime: timePtrP(base.Add(-25 * time.Second))}, + "latency_range": {MinLatency: f64PtrP(80), MaxLatency: f64PtrP(260)}, + "token_range": {MinTokens: intPtrP(100), MaxTokens: intPtrP(600)}, + "cost_range": {MinCost: f64PtrP(1.0), MaxCost: f64PtrP(2.6)}, + "missing_cost": {MissingCostOnly: true}, + "cache_direct": {CacheHitTypes: []string{"direct"}}, + "cache_semantic": {CacheHitTypes: []string{"semantic"}}, + "metadata": {MetadataFilters: map[string]string{"env": "prod"}}, + "content_search": {ContentSearch: "charlie"}, + "parent_request": {ParentRequestID: "sess1"}, + } + for name, filters := range searchCases { + t.Run("SearchLogs/"+name, func(t *testing.T) { + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + r, err := s.SearchLogs(ctx, filters, page) + if err != nil { + return nil, err + } + return searchProjection(r), nil + }) + }) + } + + t.Run("SearchLogs/sort_and_pagination", func(t *testing.T) { + // MinCost 0 keeps only non-NULL costs: NULL ordering defaults differ + // across backends and the UI always filters or sorts on populated + // columns. + filters := SearchFilters{MinCost: f64PtrP(0)} + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + r, err := s.SearchLogs(ctx, filters, PaginationOptions{Limit: 3, Offset: 1, SortBy: "cost", Order: "desc"}) + if err != nil { + return nil, err + } + return searchProjection(r), nil + }) + }) + + t.Run("GetStats", func(t *testing.T) { + for name, filters := range map[string]SearchFilters{"window": window, "team_t1": {TeamIDs: []string{"t1"}}} { + t.Run(name, func(t *testing.T) { + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + return s.GetStats(ctx, filters) + }) + }) + } + }) + + histogramCalls := map[string]func(context.Context, LogStore) (any, error){ + "requests": func(ctx context.Context, s LogStore) (any, error) { return s.GetHistogram(ctx, window, 60) }, + "tokens": func(ctx context.Context, s LogStore) (any, error) { return s.GetTokenHistogram(ctx, window, 60) }, + "cost": func(ctx context.Context, s LogStore) (any, error) { return s.GetCostHistogram(ctx, window, 60) }, + "model": func(ctx context.Context, s LogStore) (any, error) { return s.GetModelHistogram(ctx, window, 60) }, + "latency": func(ctx context.Context, s LogStore) (any, error) { return s.GetLatencyHistogram(ctx, window, 60) }, + "provider_cost": func(ctx context.Context, s LogStore) (any, error) { + return s.GetProviderCostHistogram(ctx, window, 60) + }, + "provider_tokens": func(ctx context.Context, s LogStore) (any, error) { + return s.GetProviderTokenHistogram(ctx, window, 60) + }, + "provider_latency": func(ctx context.Context, s LogStore) (any, error) { + return s.GetProviderLatencyHistogram(ctx, window, 60) + }, + "dimension_cost": func(ctx context.Context, s LogStore) (any, error) { + return s.GetDimensionCostHistogram(ctx, window, 60, DimensionProvider) + }, + "dimension_tokens": func(ctx context.Context, s LogStore) (any, error) { + return s.GetDimensionTokenHistogram(ctx, window, 60, DimensionUser) + }, + "dimension_latency": func(ctx context.Context, s LogStore) (any, error) { + return s.GetDimensionLatencyHistogram(ctx, window, 60, DimensionProvider) + }, + // Filter on the same column the SELECT aliases (SUM(cost) AS cost): + // without prefer_column_name_to_alias=1 ClickHouse resolves the WHERE + // identifier to the aggregate alias and errors (code 184). + "cost_with_cost_filter": func(ctx context.Context, s LogStore) (any, error) { + f := window + f.MinCost = f64PtrP(0.6) + return s.GetCostHistogram(ctx, f, 60) + }, + } + for name, call := range histogramCalls { + t.Run("Histogram/"+name, func(t *testing.T) { + // Latency percentiles: percentile_cont (PG) vs quantile (CH) vs + // Go-side interpolation (SQLite) agree on small exact sets; the + // loose tolerance absorbs interpolation differences without hiding + // real drift. + assertParity(t, stores, 1e-3, call) + }) + } + + t.Run("Rankings/models", func(t *testing.T) { + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + return s.GetModelRankings(ctx, window) + }) + }) + t.Run("Rankings/users", func(t *testing.T) { + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + return s.GetUserRankings(ctx, window) + }) + }) + t.Run("Rankings/dimension_virtual_key", func(t *testing.T) { + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + return s.GetDimensionRankings(ctx, window, RankingDimensionVirtualKey) + }) + }) + t.Run("Rankings/dimension_team", func(t *testing.T) { + // Fixtures carry scalar team_id only, so the Postgres fan-out's scalar + // fallback and the other backends' plain group-by must agree on the + // rankings. TotalActual/AttributedRequests are documented as + // fan-out-only (Postgres) metadata and excluded from the contract. + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + r, err := s.GetDimensionRankings(ctx, window, RankingDimensionTeam) + if err != nil { + return nil, err + } + return map[string]any{"rankings": r.Rankings, "dimension": r.Dimension}, nil + }) + }) + + t.Run("Distinct", func(t *testing.T) { + sorted := func(v []string, err error) (any, error) { + if err != nil { + return nil, err + } + sort.Strings(v) + return v, nil + } + calls := map[string]func(context.Context, LogStore) (any, error){ + "models": func(ctx context.Context, s LogStore) (any, error) { return sorted(s.GetDistinctModels(ctx, 50, "")) }, + "aliases": func(ctx context.Context, s LogStore) (any, error) { return sorted(s.GetDistinctAliases(ctx, 50, "")) }, + "routing_engines": func(ctx context.Context, s LogStore) (any, error) { + return sorted(s.GetDistinctRoutingEngines(ctx, 50, "")) + }, + "stop_reasons": func(ctx context.Context, s LogStore) (any, error) { + return sorted(s.GetDistinctStopReasons(ctx, 50, "")) + }, + "key_pairs": func(ctx context.Context, s LogStore) (any, error) { + pairs, err := s.GetDistinctKeyPairs(ctx, "virtual_key_id", "virtual_key_name", 50, "") + if err != nil { + return nil, err + } + sort.Slice(pairs, func(i, j int) bool { return pairs[i].ID < pairs[j].ID }) + return pairs, nil + }, + "metadata_keys": func(ctx context.Context, s LogStore) (any, error) { + m, err := s.GetDistinctMetadataKeys(ctx, 50, "") + if err != nil { + return nil, err + } + for k := range m { + sort.Strings(m[k]) + } + return m, nil + }, + } + for name, call := range calls { + t.Run(name, func(t *testing.T) { assertParity(t, stores, 1e-6, call) }) + } + }) + + t.Run("Sessions", func(t *testing.T) { + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + r, err := s.GetSessionLogs(ctx, "sess1", page) + if err != nil { + return nil, err + } + logs := make([]map[string]any, 0, len(r.Logs)) + for _, l := range r.Logs { + logs = append(logs, logProjection(&l)) + } + return map[string]any{"total": r.Pagination.TotalCount, "logs": logs}, nil + }) + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + return s.GetSessionSummary(ctx, "sess1") + }) + }) + + t.Run("Lookups", func(t *testing.T) { + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + l, err := s.FindByID(ctx, "p1") + if err != nil { + return nil, err + } + return logProjection(l), nil + }) + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + l, err := s.FindFirst(ctx, map[string]interface{}{"provider": "mistral"}, "id", "provider", "status") + if err != nil { + return nil, err + } + return map[string]any{"id": l.ID, "provider": l.Provider, "status": l.Status}, nil + }) + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + logs, err := s.FindAll(ctx, map[string]interface{}{"status": "success"}, "id") + if err != nil { + return nil, err + } + ids := make([]string, 0, len(logs)) + for _, l := range logs { + ids = append(ids, l.ID) + } + sort.Strings(ids) + return ids, nil + }) + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + logs, err := s.FindAllDistinct(ctx, map[string]interface{}{"status": "success"}, "provider") + if err != nil { + return nil, err + } + providers := make([]string, 0, len(logs)) + for _, l := range logs { + providers = append(providers, l.Provider) + } + sort.Strings(providers) + return providers, nil + }) + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + return s.HasLogs(ctx) + }) + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + present, err := s.IsLogEntryPresent(ctx, "p3") + if err != nil { + return nil, err + } + missing, err := s.IsLogEntryPresent(ctx, "nope") + if err != nil { + return nil, err + } + return []bool{present, missing}, nil + }) + }) + + t.Run("QueryScopeDAC", func(t *testing.T) { + // Same closure shape the enterprise DAC wrapper builds (dacscope.go + // logScope): OR-joined IN predicates over ownership columns. + scope := queryscope.QueryScope(func(db *gorm.DB) *gorm.DB { + return db.Where("(user_id IN ? OR virtual_key_id IN ?)", []string{"u1"}, []string{"vk2"}) + }) + scopedCtx := queryscope.WithQueryScope(context.Background(), scope) + refResult, err := stores[parityReference].SearchLogs(scopedCtx, window, page) + require.NoError(t, err) + refNorm := canonicalizeOrder("", jsonNormalize(t, searchProjection(refResult))) + for name, s := range stores { + if name == parityReference { + continue + } + r, err := s.SearchLogs(scopedCtx, window, page) + require.NoError(t, err, name) + var diffs []string + collectDiffs("$", refNorm, canonicalizeOrder("", jsonNormalize(t, searchProjection(r))), 1e-6, &diffs) + assert.Empty(t, diffs, "%s: DAC-scoped results must match", name) + } + + // Fail-closed parity: the no-dimension principal shape. + closed := queryscope.WithQueryScope(context.Background(), func(db *gorm.DB) *gorm.DB { + return db.Where("1 = 0") + }) + for name, s := range stores { + r, err := s.SearchLogs(closed, window, page) + require.NoError(t, err, name) + assert.Zero(t, r.Pagination.TotalCount, name) + assert.Empty(t, r.Logs, name) + } + }) + + t.Run("MCP", func(t *testing.T) { + mcpWindow := MCPToolLogSearchFilters{StartTime: timePtrP(windowStart), EndTime: timePtrP(windowEnd)} + mcpPage := PaginationOptions{Limit: 50, SortBy: "timestamp", Order: "desc"} + calls := map[string]func(context.Context, LogStore) (any, error){ + "search": func(ctx context.Context, s LogStore) (any, error) { + r, err := s.SearchMCPToolLogs(ctx, mcpWindow, mcpPage) + if err != nil { + return nil, err + } + return map[string]any{"total": r.Pagination.TotalCount, "logs": mcpProjection(r.Logs)}, nil + }, + "search_filtered": func(ctx context.Context, s LogStore) (any, error) { + r, err := s.SearchMCPToolLogs(ctx, MCPToolLogSearchFilters{ToolNames: []string{"search_web"}, Status: []string{"success", "error"}}, mcpPage) + if err != nil { + return nil, err + } + return map[string]any{"total": r.Pagination.TotalCount, "logs": mcpProjection(r.Logs)}, nil + }, + "stats": func(ctx context.Context, s LogStore) (any, error) { return s.GetMCPToolLogStats(ctx, mcpWindow) }, + "histogram": func(ctx context.Context, s LogStore) (any, error) { + return s.GetMCPHistogram(ctx, mcpWindow, 60) + }, + "cost_histogram": func(ctx context.Context, s LogStore) (any, error) { + return s.GetMCPCostHistogram(ctx, mcpWindow, 60) + }, + "top_tools": func(ctx context.Context, s LogStore) (any, error) { return s.GetMCPTopTools(ctx, mcpWindow, 10) }, + "tool_names": func(ctx context.Context, s LogStore) (any, error) { + v, err := s.GetAvailableToolNames(ctx, 50, "") + if err != nil { + return nil, err + } + sort.Strings(v) + return v, nil + }, + "server_labels": func(ctx context.Context, s LogStore) (any, error) { + v, err := s.GetAvailableServerLabels(ctx, 50, "") + if err != nil { + return nil, err + } + sort.Strings(v) + return v, nil + }, + "virtual_keys": func(ctx context.Context, s LogStore) (any, error) { + logs, err := s.GetAvailableMCPVirtualKeys(ctx, 50, "") + if err != nil { + return nil, err + } + type vk struct{ ID, Name string } + out := make([]vk, 0, len(logs)) + for _, l := range logs { + pair := vk{} + if l.VirtualKeyID != nil { + pair.ID = *l.VirtualKeyID + } + if l.VirtualKeyName != nil { + pair.Name = *l.VirtualKeyName + } + out = append(out, pair) + } + sort.Slice(out, func(i, j int) bool { return out[i].ID < out[j].ID }) + return out, nil + }, + "find": func(ctx context.Context, s LogStore) (any, error) { + l, err := s.FindMCPToolLog(ctx, "m1") + if err != nil { + return nil, err + } + return mcpProjection([]MCPToolLog{*l}), nil + }, + "has_logs": func(ctx context.Context, s LogStore) (any, error) { return s.HasMCPToolLogs(ctx) }, + } + for name, call := range calls { + t.Run(name, func(t *testing.T) { assertParity(t, stores, 1e-3, call) }) + } + }) + + t.Run("NodeUsage", func(t *testing.T) { + // inc_number is Postgres-assigned and NULL elsewhere, so the cursor's + // IncNumber field is excluded; everything the budget reconciler + // consumes must match. + project := func(a *NodeUsageAggregate) any { + return map[string]any{ + "budget_costs": a.BudgetCosts, "rl_requests": a.RateLimitRequests, + "rl_tokens": a.RateLimitTokens, "rows": a.RowCount, + "max_ts_ms": a.MaxTimestamp.UnixMilli(), "max_log_id": a.MaxLogID, + } + } + start := NodeUsageCursor{Timestamp: base.Add(-time.Hour)} + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + a, err := s.GetNodeUsageAfter(ctx, "pnode", start) + if err != nil { + return nil, err + } + return project(a), nil + }) + // Cursor advance parity: a second scan from each store's own returned + // cursor must aggregate nothing on every backend. + for name, s := range stores { + first, err := s.GetNodeUsageAfter(ctx, "pnode", start) + require.NoError(t, err, name) + second, err := s.GetNodeUsageAfter(ctx, "pnode", first.NextCursor) + require.NoError(t, err, name) + assert.Zero(t, second.RowCount, "%s: cursor must not re-aggregate", name) + } + }) + + // --- Phase: mutations --- + + t.Run("Mutations", func(t *testing.T) { + t.Run("update_map", func(t *testing.T) { + runOnAll(t, stores, func(ctx context.Context, s LogStore) error { + return s.Update(ctx, "p6", map[string]interface{}{ + "status": "error", "latency": 99.0, "stop_reason": "content_filter", + }) + }) + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + l, err := s.FindByID(ctx, "p6") + if err != nil { + return nil, err + } + return logProjection(l), nil + }) + }) + t.Run("update_struct", func(t *testing.T) { + // Struct updates write non-zero fields only, on every backend. + runOnAll(t, stores, func(ctx context.Context, s LogStore) error { + return s.Update(ctx, "p8", &Log{Model: "gpt-4o-turbo", Cost: f64PtrP(0.85)}) + }) + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + l, err := s.FindByID(ctx, "p8") + if err != nil { + return nil, err + } + return logProjection(l), nil + }) + }) + t.Run("update_missing_row", func(t *testing.T) { + for name, s := range stores { + err := s.Update(ctx, "missing-id", map[string]interface{}{"status": "success"}) + assert.ErrorIs(t, err, ErrNotFound, name) + } + }) + t.Run("create_if_not_exists_keeps_existing", func(t *testing.T) { + // The retried "processing" insert after an update must be a no-op + // on every backend (ClickHouse would be last-write-wins without + // its existence check). + runOnAll(t, stores, func(ctx context.Context, s LogStore) error { + return s.CreateIfNotExists(ctx, paritySpecs()[5].toLog(base)) + }) + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + l, err := s.FindByID(ctx, "p6") + if err != nil { + return nil, err + } + return logProjection(l), nil + }) + }) + t.Run("bulk_update_cost", func(t *testing.T) { + runOnAll(t, stores, func(ctx context.Context, s LogStore) error { + return s.BulkUpdateCost(ctx, map[string]float64{"p1": 0.9, "p4": 2.9}) + }) + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + var costs []map[string]any + for _, id := range []string{"p1", "p4"} { + l, err := s.FindByID(ctx, id) + if err != nil { + return nil, err + } + costs = append(costs, map[string]any{"id": l.ID, "cost": l.Cost}) + } + return costs, nil + }) + }) + t.Run("update_mcp_map_and_struct", func(t *testing.T) { + runOnAll(t, stores, func(ctx context.Context, s LogStore) error { + if err := s.UpdateMCPToolLog(ctx, "m3", map[string]interface{}{"status": "error", "latency": 45.0}); err != nil { + return err + } + return s.UpdateMCPToolLog(ctx, "m2", &MCPToolLog{Status: "success", Cost: f64PtrP(0.02)}) + }) + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + var out []map[string]any + for _, id := range []string{"m2", "m3"} { + l, err := s.FindMCPToolLog(ctx, id) + if err != nil { + return nil, err + } + out = append(out, mcpProjection([]MCPToolLog{*l})...) + } + return out, nil + }) + for name, s := range stores { + err := s.UpdateMCPToolLog(ctx, "missing-id", map[string]interface{}{"status": "success"}) + assert.ErrorIs(t, err, ErrNotFound, name) + } + }) + }) + + // --- Phase: async jobs --- + + t.Run("AsyncJobs", func(t *testing.T) { + mkJob := func(id string, status string, createdOffset time.Duration, expires *time.Time) *AsyncJob { + return &AsyncJob{ + ID: id, Status: schemas.AsyncJobStatus(status), RequestType: schemas.RequestType("chat"), + Response: `{"ok":true}`, ResultTTL: 3600, ExpiresAt: expires, + CreatedAt: base.Add(createdOffset), + } + } + runOnAll(t, stores, func(ctx context.Context, s LogStore) error { + if err := s.CreateAsyncJob(ctx, mkJob("j1", "processing", -time.Minute, nil)); err != nil { + return err + } + // j2 expired ten minutes ago; j3 is a stale processing job. + if err := s.CreateAsyncJob(ctx, mkJob("j2", "completed", -time.Hour, timePtrP(base.Add(-10*time.Minute)))); err != nil { + return err + } + return s.CreateAsyncJob(ctx, mkJob("j3", "processing", -2*time.Hour, nil)) + }) + + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + j, err := s.FindAsyncJobByID(ctx, "j1") + if err != nil { + return nil, err + } + return asyncJobProjection(j), nil + }) + // Expired jobs are invisible on every backend. + for name, s := range stores { + _, err := s.FindAsyncJobByID(ctx, "j2") + assert.ErrorIs(t, err, ErrNotFound, name) + } + + runOnAll(t, stores, func(ctx context.Context, s LogStore) error { + return s.UpdateAsyncJob(ctx, "j1", map[string]interface{}{ + "status": "completed", "status_code": 200, "response": `{"done":true}`, + }) + }) + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + j, err := s.FindAsyncJobByID(ctx, "j1") + if err != nil { + return nil, err + } + return asyncJobProjection(j), nil + }) + + // Stale cleanup removes j3 (processing, created 2h ago) and reports + // the same count everywhere. + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + return s.DeleteStaleAsyncJobs(ctx, base.Add(-time.Hour)) + }) + // Expired cleanup removes j2 with the same count everywhere. + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + return s.DeleteExpiredAsyncJobs(ctx) + }) + }) + + // --- Phase: deletes (destructive; order matters) --- + + t.Run("Deletes", func(t *testing.T) { + t.Run("flush_processing", func(t *testing.T) { + // Flush drops processing rows older than since: p5 (base-60s). + runOnAll(t, stores, func(ctx context.Context, s LogStore) error { + return s.Flush(ctx, base.Add(-55*time.Second)) + }) + assertParity(t, stores, 1e-6, remainingLogIDs) + }) + t.Run("delete_log", func(t *testing.T) { + runOnAll(t, stores, func(ctx context.Context, s LogStore) error { + return s.DeleteLog(ctx, "p1") + }) + for name, s := range stores { + _, err := s.FindByID(ctx, "p1") + assert.ErrorIs(t, err, ErrNotFound, name) + } + assertParity(t, stores, 1e-6, remainingLogIDs) + }) + t.Run("delete_logs", func(t *testing.T) { + runOnAll(t, stores, func(ctx context.Context, s LogStore) error { + return s.DeleteLogs(ctx, []string{"p2", "p3"}) + }) + assertParity(t, stores, 1e-6, remainingLogIDs) + }) + t.Run("delete_logs_batch", func(t *testing.T) { + // Deletes rows created before base-45s (p4, p6 remain from that + // window) and must report the same count on every backend + // (ClickHouse mutations report 0 rows affected natively). + assertParity(t, stores, 1e-6, func(ctx context.Context, s LogStore) (any, error) { + return s.DeleteLogsBatch(ctx, base.Add(-45*time.Second), 100) + }) + assertParity(t, stores, 1e-6, remainingLogIDs) + }) + t.Run("flush_mcp", func(t *testing.T) { + // m4 is the only processing MCP row; FlushMCPToolLogs drops it. + runOnAll(t, stores, func(ctx context.Context, s LogStore) error { + return s.FlushMCPToolLogs(ctx, base) + }) + assertParity(t, stores, 1e-6, remainingMCPIDs) + }) + t.Run("delete_mcp", func(t *testing.T) { + runOnAll(t, stores, func(ctx context.Context, s LogStore) error { + return s.DeleteMCPToolLogs(ctx, []string{"m1"}) + }) + for name, s := range stores { + _, err := s.FindMCPToolLog(ctx, "m1") + assert.ErrorIs(t, err, ErrNotFound, name) + } + assertParity(t, stores, 1e-6, remainingMCPIDs) + }) + }) + + // --- Phase: shutdown --- + + t.Run("Close", func(t *testing.T) { + runOnAll(t, stores, func(ctx context.Context, s LogStore) error { + return s.Close(ctx) + }) + }) +} diff --git a/framework/logstore/matviews.go b/framework/logstore/matviews.go index 9331101a2ee..c98befe30fa 100644 --- a/framework/logstore/matviews.go +++ b/framework/logstore/matviews.go @@ -41,6 +41,7 @@ SELECT COUNT(*) AS count, SUM(CASE WHEN status = 'success' THEN 1 ELSE 0 END) AS success_count, SUM(CASE WHEN status = 'error' THEN 1 ELSE 0 END) AS error_count, + SUM(CASE WHEN status = 'cancelled' THEN 1 ELSE 0 END) AS cancelled_count, COALESCE(AVG(latency), 0) AS avg_latency, COALESCE(percentile_cont(0.90) WITHIN GROUP (ORDER BY latency), 0) AS p90_latency, COALESCE(percentile_cont(0.95) WITHIN GROUP (ORDER BY latency), 0) AS p95_latency, @@ -51,7 +52,7 @@ SELECT COALESCE(SUM(cached_read_tokens), 0) AS total_cached_read_tokens, COALESCE(SUM(cost), 0) AS total_cost FROM logs -WHERE status IN ('success', 'error') +WHERE status IN ('success', 'error', 'cancelled') GROUP BY 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13 ` @@ -80,6 +81,7 @@ var mvLogsHourlyRequiredColumns = []string{ "customer_id", "business_unit_id", "alias", + "cancelled_count", } // legacyMatViewNames are matviews from previous schema versions that no longer @@ -764,6 +766,7 @@ func startMatViewRefresher(ctx context.Context, db *gorm.DB, interval time.Durat func canUseMatViewFilters(f SearchFilters) bool { return f.ContentSearch == "" && len(f.MetadataFilters) == 0 && + canUseMatViewStatusFilter(f.Status) && len(f.RoutingEngineUsed) == 0 && len(f.StopReasons) == 0 && f.MinLatency == nil && f.MaxLatency == nil && @@ -776,6 +779,24 @@ func canUseMatViewFilters(f SearchFilters) bool { len(f.CustomerIDs) == 0 } +func canUseMatViewStatusFilter(statuses []string) bool { + for _, status := range statuses { + if !isTerminalLogStatus(status) { + return false + } + } + return true +} + +func isTerminalLogStatus(status string) bool { + for _, terminalStatus := range terminalLogStatuses { + if status == terminalStatus { + return true + } + } + return false +} + // canUseMatView checks both that materialized views are ready (created and // populated) and that the given filters are eligible for the matview path. // This prevents queries from hitting non-existent views during the startup @@ -950,7 +971,7 @@ func (s *RDBLogStore) getStatsFromMatView(ctx context.Context, filters SearchFil alignedEnd := filters.EndTime.Truncate(time.Hour).Add(time.Hour - time.Nanosecond) alignedFilters.EndTime = &alignedEnd } - cacheBase := s.ScopedDB(ctx).Model(&Log{}).Where("status IN ?", []string{"success", "error"}) + cacheBase := s.ScopedDB(ctx).Model(&Log{}).Where("status IN ?", terminalLogStatuses) direct, semantic, err := s.aggregateCacheHits(ctx, cacheBase, alignedFilters) if err != nil { s.logger.Warn(fmt.Sprintf("logstore: failed to aggregate cache-hit stats, skipping: %s", err)) @@ -963,13 +984,14 @@ func (s *RDBLogStore) getStatsFromMatView(ctx context.Context, filters SearchFil } // getHistogramFromMatView returns time-bucketed request counts (total, -// success, error) by re-aggregating hourly buckets from mv_logs_hourly. +// success, error, cancelled) by re-aggregating hourly buckets from mv_logs_hourly. func (s *RDBLogStore) getHistogramFromMatView(ctx context.Context, filters SearchFilters, bucketSizeSeconds int64) (*HistogramResult, error) { var results []struct { BucketTimestamp int64 `gorm:"column:bucket_timestamp"` Total int64 `gorm:"column:total"` Success int64 `gorm:"column:success"` ErrorCount int64 `gorm:"column:error_count"` + CancelledCount int64 `gorm:"column:cancelled_count"` } q := s.ScopedDB(ctx).Table("mv_logs_hourly") q = s.applyMatViewFilters(q, filters) @@ -977,7 +999,8 @@ func (s *RDBLogStore) getHistogramFromMatView(ctx context.Context, filters Searc CAST(FLOOR(EXTRACT(EPOCH FROM hour) / %d) * %d AS BIGINT) AS bucket_timestamp, SUM(count) AS total, SUM(success_count) AS success, - SUM(error_count) AS error_count + SUM(error_count) AS error_count, + SUM(cancelled_count) AS cancelled_count `, bucketSizeSeconds, bucketSizeSeconds)). Group("bucket_timestamp"). Order("bucket_timestamp ASC"). @@ -985,9 +1008,9 @@ func (s *RDBLogStore) getHistogramFromMatView(ctx context.Context, filters Searc return nil, err } - resultMap := make(map[int64]*struct{ total, success, errCount int64 }, len(results)) + resultMap := make(map[int64]*struct{ total, success, errCount, cancelledCount int64 }, len(results)) for _, r := range results { - resultMap[r.BucketTimestamp] = &struct{ total, success, errCount int64 }{r.Total, r.Success, r.ErrorCount} + resultMap[r.BucketTimestamp] = &struct{ total, success, errCount, cancelledCount int64 }{r.Total, r.Success, r.ErrorCount, r.CancelledCount} } allTimestamps := generateBucketTimestamps(filters.StartTime, filters.EndTime, bucketSizeSeconds) @@ -998,6 +1021,7 @@ func (s *RDBLogStore) getHistogramFromMatView(ctx context.Context, filters Searc b.Count = a.total b.Success = a.success b.Error = a.errCount + b.Cancelled = a.cancelledCount } buckets = append(buckets, b) } @@ -1104,7 +1128,7 @@ func (s *RDBLogStore) getCostHistogramFromMatView(ctx context.Context, filters S } // getModelHistogramFromMatView returns time-bucketed model usage with -// success/error breakdown per model from mv_logs_hourly. +// success/error/cancelled breakdown per model from mv_logs_hourly. func (s *RDBLogStore) getModelHistogramFromMatView(ctx context.Context, filters SearchFilters, bucketSizeSeconds int64) (*ModelHistogramResult, error) { var results []struct { BucketTimestamp int64 `gorm:"column:bucket_timestamp"` @@ -1112,6 +1136,7 @@ func (s *RDBLogStore) getModelHistogramFromMatView(ctx context.Context, filters Total int64 `gorm:"column:total"` Success int64 `gorm:"column:success"` ErrorCount int64 `gorm:"column:error_count"` + CancelledCount int64 `gorm:"column:cancelled_count"` } q := s.ScopedDB(ctx).Table("mv_logs_hourly") q = s.applyMatViewFilters(q, filters) @@ -1120,7 +1145,8 @@ func (s *RDBLogStore) getModelHistogramFromMatView(ctx context.Context, filters model, SUM(count) AS total, SUM(success_count) AS success, - SUM(error_count) AS error_count + SUM(error_count) AS error_count, + SUM(cancelled_count) AS cancelled_count `, bucketSizeSeconds, bucketSizeSeconds)). Group("bucket_timestamp, model"). Order("bucket_timestamp ASC"). @@ -1143,6 +1169,7 @@ func (s *RDBLogStore) getModelHistogramFromMatView(ctx context.Context, filters existing.Total += r.Total existing.Success += r.Success existing.Error += r.ErrorCount + existing.Cancelled += r.CancelledCount a.byModel[r.Model] = existing modelsSet[r.Model] = struct{}{} } diff --git a/framework/logstore/multi_team_filter_test.go b/framework/logstore/multi_team_filter_test.go index 8127fb88a79..7c4ebb08eb0 100644 --- a/framework/logstore/multi_team_filter_test.go +++ b/framework/logstore/multi_team_filter_test.go @@ -82,7 +82,9 @@ func TestTeamOrBUFanoutFrom(t *testing.T) { func TestCanUseMatViewFilters_ExcludesTeamBU(t *testing.T) { assert.True(t, canUseMatViewFilters(SearchFilters{}), "empty filters → matview eligible") assert.True(t, canUseMatViewFilters(SearchFilters{Providers: []string{"openai"}}), "provider filter stays matview-eligible") + assert.True(t, canUseMatViewFilters(SearchFilters{Status: []string{"cancelled"}}), "cancelled is materialized as a terminal status") + assert.False(t, canUseMatViewFilters(SearchFilters{Status: []string{"processing"}}), "processing is not present in mv_logs_hourly") assert.False(t, canUseMatViewFilters(SearchFilters{TeamIDs: []string{"t1"}}), "team filter must force the raw path") assert.False(t, canUseMatViewFilters(SearchFilters{BusinessUnitIDs: []string{"bu1"}}), "BU filter must force the raw path") assert.False(t, canUseMatViewFilters(SearchFilters{CustomerIDs: []string{"c1"}}), "customer filter must force the raw path") diff --git a/framework/logstore/rdb.go b/framework/logstore/rdb.go index 34312b3b2ea..420e372128d 100644 --- a/framework/logstore/rdb.go +++ b/framework/logstore/rdb.go @@ -44,6 +44,8 @@ const ( defaultFilterDataCutoffDays = 30 ) +var terminalLogStatuses = []string{"success", "error", "cancelled"} + // RDBLogStore represents a log store that uses a SQLite database. type RDBLogStore struct { db *gorm.DB @@ -288,13 +290,19 @@ func (s *RDBLogStore) applyFilters(baseQuery *gorm.DB, filters SearchFilters) *g } } if len(valid) > 0 { - if s.db.Dialector.Name() == "postgres" { + switch s.db.Dialector.Name() { + case "postgres": // Match the same loose-JSON guard used by aggregateCacheHits so the regex extract is safe. baseQuery = baseQuery.Where( "cache_debug IS NOT NULL AND cache_debug <> '' AND cache_debug ~ '^\\s*\\{.*\\}\\s*$' AND substring(cache_debug from '\"hit_type\"[[:space:]]*:[[:space:]]*\"([^\"]+)\"') IN ?", valid, ) - } else { + case "clickhouse": + baseQuery = baseQuery.Where( + "cache_debug IS NOT NULL AND cache_debug != '' AND isValidJSON(cache_debug) AND JSONExtractString(cache_debug, 'hit_type') IN ?", + valid, + ) + default: baseQuery = baseQuery.Where( "cache_debug IS NOT NULL AND cache_debug != '' AND json_valid(cache_debug) AND json_extract(cache_debug, '$.hit_type') IN ?", valid, @@ -316,9 +324,12 @@ func (s *RDBLogStore) applyFilters(baseQuery *gorm.DB, filters SearchFilters) *g dialect := s.db.Dialector.Name() // Guard must match the partial-index predicate so the planner uses the GIN index. // SQLite does not support IS JSON OBJECT, so fall back to the equivalent json_type check. - if dialect == "postgres" { + switch dialect { + case "postgres": baseQuery = baseQuery.Where("metadata IS NOT NULL AND metadata IS JSON OBJECT") - } else { + case "clickhouse": + baseQuery = baseQuery.Where("metadata IS NOT NULL AND isValidJSON(metadata)") + default: baseQuery = baseQuery.Where("metadata IS NOT NULL AND json_valid(metadata) AND json_type(metadata) = 'object'") } for key, value := range filters.MetadataFilters { @@ -332,6 +343,10 @@ func (s *RDBLogStore) applyFilters(baseQuery *gorm.DB, filters SearchFilters) *g // strings — always match as a string to avoid type mismatch with jsonb. jsonFragment := fmt.Sprintf(`{%q: %q}`, key, value) baseQuery = baseQuery.Where("metadata::jsonb @> ?::jsonb", jsonFragment) + case "clickhouse": + // Metadata values are stored as JSON strings (see postgres note); + // match them as strings via JSONExtractString. + baseQuery = baseQuery.Where("JSONExtractString(metadata, ?) = ?", key, value) default: // SQLite: quote the member name so dots/hyphens stay part of the key path := `$."` + key + `"` @@ -459,9 +474,23 @@ func (s *RDBLogStore) GetNodeUsageAfter(ctx context.Context, nodeID string, curs Where("cluster_node_id = ?", nodeID). Where("status = ?", "success") orderBy := "timestamp ASC, id ASC" + clickhouse := s.db.Dialector.Name() == "clickhouse" if cursor.IncNumber != nil { query = query.Where("inc_number > ?", *cursor.IncNumber) orderBy = "inc_number ASC" + } else if clickhouse { + // The GORM ClickHouse driver truncates time.Time args to whole seconds + // (toDateTime('...')), which would rewind the cursor to the start of its + // second and re-aggregate rows already counted on the previous scan. + // Bind epoch millis instead; DateTime64(3) storage makes this exact. + // inc_number stays NULL on ClickHouse (no DB-assigned autoincrement), so + // this cursor form is the steady state, not just the first scan. + tsMs := cursor.Timestamp.UnixMilli() + if cursor.LogID != "" { + query = query.Where("timestamp > fromUnixTimestamp64Milli(?) OR (timestamp = fromUnixTimestamp64Milli(?) AND id > ?)", tsMs, tsMs, cursor.LogID) + } else { + query = query.Where("timestamp > fromUnixTimestamp64Milli(?)", tsMs) + } } else if cursor.LogID != "" { query = query.Where("timestamp > ? OR (timestamp = ? AND id > ?)", cursor.Timestamp, cursor.Timestamp, cursor.LogID) } else { @@ -863,8 +892,9 @@ func (s *RDBLogStore) listSelectColumns() string { "selected_key_id", "selected_key_name", "virtual_key_id", "virtual_key_name", "routing_engines_used", "routing_rule_id", "routing_rule_name", - "user_id", "team_id", "team_name", "customer_id", "customer_name", + "user_id", "user_name", "team_id", "team_name", "customer_id", "customer_name", "business_unit_id", "business_unit_name", + "team_ids", "team_names", "customer_ids", "customer_names", "business_unit_ids", "business_unit_names", "speech_input", "transcription_input", "image_generation_input", "video_generation_input", "latency", "token_usage", "cost", "status", "error_details", "stream", fmt.Sprintf("substr(content_summary, 1, %d) AS content_summary", maxContentSummaryBytes), @@ -891,6 +921,14 @@ func (s *RDBLogStore) listSelectColumns() string { ELSE bifrost_safe_jsonb(responses_input_history) END AS responses_input_history` outputMessageExpr = `CASE WHEN object_type = 'realtime.turn' THEN output_message ELSE NULL END AS output_message` + case "clickhouse": + // ClickHouse: return the full history columns as-is. The last-message + // truncation optimization the SQLite/Postgres list path applies is + // deferred (correctness over payload size); hybrid offloading and + // content_summary already bound list payloads in practice. + inputHistoryExpr = `input_history AS input_history` + responsesInputExpr = `responses_input_history AS responses_input_history` + outputMessageExpr = `CASE WHEN object_type = 'realtime.turn' THEN output_message ELSE NULL END AS output_message` default: // sqlite inputHistoryExpr = `CASE WHEN object_type = 'realtime.turn' THEN input_history @@ -937,7 +975,7 @@ func (s *RDBLogStore) GetStats(ctx context.Context, filters SearchFilters) (*Sea } if totalCount > 0 { - // Single query for all completed-request stats: counts, latency, tokens, cost + // Single query for all terminal-request stats: counts, latency, tokens, cost var result struct { CompletedCount sql.NullInt64 `gorm:"column:completed_count"` SuccessCount sql.NullInt64 `gorm:"column:success_count"` @@ -948,7 +986,7 @@ func (s *RDBLogStore) GetStats(ctx context.Context, filters SearchFilters) (*Sea statsQuery := s.ScopedDB(ctx).Model(&Log{}) statsQuery = s.applyFilters(statsQuery, filters) - statsQuery = statsQuery.Where("status IN ?", []string{"success", "error"}) + statsQuery = statsQuery.Where("status IN ?", terminalLogStatuses) if err := statsQuery.Select(` COUNT(*) as completed_count, @@ -991,7 +1029,7 @@ func (s *RDBLogStore) GetStats(ctx context.Context, filters SearchFilters) (*Sea userFacingQuery = s.applyFilters(userFacingQuery, filters) // Scope to root rows only so denominator and numerator are drawn from the same population. // A chain is successful if the root itself succeeded or any of its fallbacks succeeded. - userFacingQuery = userFacingQuery.Where("fallback_index = ?", 0).Where("status IN ?", []string{"success", "error"}) + userFacingQuery = userFacingQuery.Where("fallback_index = ?", 0).Where("status IN ?", terminalLogStatuses) // Use a LEFT JOIN instead of a correlated EXISTS subquery: the inner set is computed // once and hash-joined, reducing complexity from O(N×M) to O(N+M). // The inner subquery is bounded by the same time window as the outer query for @@ -1030,7 +1068,7 @@ func (s *RDBLogStore) GetStats(ctx context.Context, filters SearchFilters) (*Sea } // Count cache hits by hit_type from cache_debug JSON - cacheBase := s.ScopedDB(ctx).Model(&Log{}).Where("status IN ?", []string{"success", "error"}) + cacheBase := s.ScopedDB(ctx).Model(&Log{}).Where("status IN ?", terminalLogStatuses) direct, semantic, err := s.aggregateCacheHits(ctx, cacheBase, filters) if err != nil { s.logger.Warn(fmt.Sprintf("logstore: failed to aggregate cache-hit stats, skipping: %s", err)) @@ -1050,7 +1088,8 @@ func (s *RDBLogStore) aggregateCacheHits(ctx context.Context, base *gorm.DB, fil SemanticHits sql.NullInt64 `gorm:"column:semantic_hits"` } q := s.applyFilters(base, filters) - if s.db.Dialector.Name() == "postgres" { + switch s.db.Dialector.Name() { + case "postgres": q = q.Where("cache_debug IS NOT NULL AND cache_debug <> '' AND cache_debug ~ '^\\s*\\{.*\\}\\s*$'") if err := q.Select( `SUM(CASE WHEN substring(cache_debug from '"hit_type"[[:space:]]*:[[:space:]]*"([^"]+)"') = 'direct' THEN 1 ELSE 0 END) AS direct_hits, ` + @@ -1058,7 +1097,15 @@ func (s *RDBLogStore) aggregateCacheHits(ctx context.Context, base *gorm.DB, fil ).Scan(&result).Error; err != nil { return nil, nil, fmt.Errorf("failed to aggregate cache-hit stats: %w", err) } - } else { + case "clickhouse": + q = q.Where("cache_debug IS NOT NULL AND cache_debug != '' AND isValidJSON(cache_debug)") + if err := q.Select( + `SUM(CASE WHEN JSONExtractString(cache_debug, 'hit_type') = 'direct' THEN 1 ELSE 0 END) AS direct_hits, ` + + `SUM(CASE WHEN JSONExtractString(cache_debug, 'hit_type') = 'semantic' THEN 1 ELSE 0 END) AS semantic_hits`, + ).Scan(&result).Error; err != nil { + return nil, nil, fmt.Errorf("failed to aggregate cache-hit stats: %w", err) + } + default: q = q.Where("cache_debug IS NOT NULL AND cache_debug != '' AND json_valid(cache_debug)") if err := q.Select( `SUM(CASE WHEN json_extract(cache_debug, '$.hit_type') = 'direct' THEN 1 ELSE 0 END) AS direct_hits, ` + @@ -1090,7 +1137,7 @@ func (s *RDBLogStore) GetHistogram(ctx context.Context, filters SearchFilters, b // Build query with filters baseQuery := s.ScopedDB(ctx).Model(&Log{}) baseQuery = s.applyFilters(baseQuery, filters) - baseQuery = baseQuery.Where("status IN ?", []string{"success", "error"}) + baseQuery = baseQuery.Where("status IN ?", terminalLogStatuses) // Query for histogram buckets - use int64 for bucket timestamp to avoid parsing issues var results []struct { @@ -1098,36 +1145,17 @@ func (s *RDBLogStore) GetHistogram(ctx context.Context, filters SearchFilters, b Total int64 `gorm:"column:total"` Success int64 `gorm:"column:success"` Error int64 `gorm:"column:error_count"` + Cancelled int64 `gorm:"column:cancelled_count"` } // Build select clause with database-specific unix timestamp calculation - var selectClause string - switch dialect { - case "sqlite": - // SQLite: use strftime to get unix timestamp, then bucket - selectClause = fmt.Sprintf(` - (CAST(strftime('%%s', timestamp) AS INTEGER) / %d) * %d as bucket_timestamp, - COUNT(*) as total, - SUM(CASE WHEN status = 'success' THEN 1 ELSE 0 END) as success, - SUM(CASE WHEN status = 'error' THEN 1 ELSE 0 END) as error_count - `, bucketSizeSeconds, bucketSizeSeconds) - case "mysql": - // MySQL: use UNIX_TIMESTAMP - selectClause = fmt.Sprintf(` - (FLOOR(UNIX_TIMESTAMP(timestamp) / %d) * %d) as bucket_timestamp, - COUNT(*) as total, - SUM(CASE WHEN status = 'success' THEN 1 ELSE 0 END) as success, - SUM(CASE WHEN status = 'error' THEN 1 ELSE 0 END) as error_count - `, bucketSizeSeconds, bucketSizeSeconds) - default: - // PostgreSQL (and others): use EXTRACT(EPOCH FROM timestamp) - selectClause = fmt.Sprintf(` - CAST(FLOOR(EXTRACT(EPOCH FROM timestamp) / %d) * %d AS BIGINT) as bucket_timestamp, + selectClause := fmt.Sprintf(` + %s as bucket_timestamp, COUNT(*) as total, SUM(CASE WHEN status = 'success' THEN 1 ELSE 0 END) as success, - SUM(CASE WHEN status = 'error' THEN 1 ELSE 0 END) as error_count - `, bucketSizeSeconds, bucketSizeSeconds) - } + SUM(CASE WHEN status = 'error' THEN 1 ELSE 0 END) as error_count, + SUM(CASE WHEN status = 'cancelled' THEN 1 ELSE 0 END) as cancelled_count + `, unixBucketExpr(dialect, bucketSizeSeconds)) if err := baseQuery. Select(selectClause). @@ -1139,19 +1167,22 @@ func (s *RDBLogStore) GetHistogram(ctx context.Context, filters SearchFilters, b // Create a map of bucket timestamp -> result for quick lookup resultMap := make(map[int64]struct { - Total int64 - Success int64 - Error int64 + Total int64 + Success int64 + Error int64 + Cancelled int64 }) for _, r := range results { resultMap[r.BucketTimestamp] = struct { - Total int64 - Success int64 - Error int64 + Total int64 + Success int64 + Error int64 + Cancelled int64 }{ - Total: r.Total, - Success: r.Success, - Error: r.Error, + Total: r.Total, + Success: r.Success, + Error: r.Error, + Cancelled: r.Cancelled, } } @@ -1167,6 +1198,7 @@ func (s *RDBLogStore) GetHistogram(ctx context.Context, filters SearchFilters, b Count: r.Total, Success: r.Success, Error: r.Error, + Cancelled: r.Cancelled, } } return &HistogramResult{ @@ -1184,6 +1216,7 @@ func (s *RDBLogStore) GetHistogram(ctx context.Context, filters SearchFilters, b Count: data.Total, Success: data.Success, Error: data.Error, + Cancelled: data.Cancelled, } } else { buckets[i] = HistogramBucket{ @@ -1214,8 +1247,8 @@ func (s *RDBLogStore) GetTokenHistogram(ctx context.Context, filters SearchFilte baseQuery := s.ScopedDB(ctx).Model(&Log{}) baseQuery = s.applyFilters(baseQuery, filters) - // Only count completed requests for token stats - baseQuery = baseQuery.Where("status IN ?", []string{"success", "error"}) + // Only count terminal requests for token stats + baseQuery = baseQuery.Where("status IN ?", terminalLogStatuses) var results []struct { BucketTimestamp int64 `gorm:"column:bucket_timestamp"` @@ -1225,33 +1258,13 @@ func (s *RDBLogStore) GetTokenHistogram(ctx context.Context, filters SearchFilte CachedReadTokens int64 `gorm:"column:cached_read_tokens"` } - var selectClause string - switch dialect { - case "sqlite": - selectClause = fmt.Sprintf(` - (CAST(strftime('%%s', timestamp) AS INTEGER) / %d) * %d as bucket_timestamp, - COALESCE(SUM(prompt_tokens), 0) as prompt_tokens, - COALESCE(SUM(completion_tokens), 0) as completion_tokens, - COALESCE(SUM(total_tokens), 0) as total_tokens, - COALESCE(SUM(cached_read_tokens), 0) as cached_read_tokens - `, bucketSizeSeconds, bucketSizeSeconds) - case "mysql": - selectClause = fmt.Sprintf(` - (FLOOR(UNIX_TIMESTAMP(timestamp) / %d) * %d) as bucket_timestamp, - COALESCE(SUM(prompt_tokens), 0) as prompt_tokens, - COALESCE(SUM(completion_tokens), 0) as completion_tokens, - COALESCE(SUM(total_tokens), 0) as total_tokens, - COALESCE(SUM(cached_read_tokens), 0) as cached_read_tokens - `, bucketSizeSeconds, bucketSizeSeconds) - default: - selectClause = fmt.Sprintf(` - CAST(FLOOR(EXTRACT(EPOCH FROM timestamp) / %d) * %d AS BIGINT) as bucket_timestamp, + selectClause := fmt.Sprintf(` + %s as bucket_timestamp, COALESCE(SUM(prompt_tokens), 0) as prompt_tokens, COALESCE(SUM(completion_tokens), 0) as completion_tokens, COALESCE(SUM(total_tokens), 0) as total_tokens, COALESCE(SUM(cached_read_tokens), 0) as cached_read_tokens - `, bucketSizeSeconds, bucketSizeSeconds) - } + `, unixBucketExpr(dialect, bucketSizeSeconds)) if err := baseQuery. Select(selectClause). @@ -1340,8 +1353,8 @@ func (s *RDBLogStore) GetCostHistogram(ctx context.Context, filters SearchFilter baseQuery := s.ScopedDB(ctx).Model(&Log{}) baseQuery = s.applyFilters(baseQuery, filters) - // Only count completed requests with cost - baseQuery = baseQuery.Where("status IN ?", []string{"success", "error"}) + // Only count terminal requests with cost + baseQuery = baseQuery.Where("status IN ?", terminalLogStatuses) baseQuery = baseQuery.Where("cost IS NOT NULL AND cost > 0") // Query grouped by bucket and model @@ -1351,27 +1364,11 @@ func (s *RDBLogStore) GetCostHistogram(ctx context.Context, filters SearchFilter TotalCost float64 `gorm:"column:total_cost"` } - var selectClause string - switch dialect { - case "sqlite": - selectClause = fmt.Sprintf(` - (CAST(strftime('%%s', timestamp) AS INTEGER) / %d) * %d as bucket_timestamp, - model, - COALESCE(SUM(cost), 0) as total_cost - `, bucketSizeSeconds, bucketSizeSeconds) - case "mysql": - selectClause = fmt.Sprintf(` - (FLOOR(UNIX_TIMESTAMP(timestamp) / %d) * %d) as bucket_timestamp, - model, - COALESCE(SUM(cost), 0) as total_cost - `, bucketSizeSeconds, bucketSizeSeconds) - default: - selectClause = fmt.Sprintf(` - CAST(FLOOR(EXTRACT(EPOCH FROM timestamp) / %d) * %d AS BIGINT) as bucket_timestamp, + selectClause := fmt.Sprintf(` + %s as bucket_timestamp, model, COALESCE(SUM(cost), 0) as total_cost - `, bucketSizeSeconds, bucketSizeSeconds) - } + `, unixBucketExpr(dialect, bucketSizeSeconds)) if err := baseQuery. Select(selectClause). @@ -1462,7 +1459,7 @@ func (s *RDBLogStore) GetModelHistogram(ctx context.Context, filters SearchFilte baseQuery := s.ScopedDB(ctx).Model(&Log{}) baseQuery = s.applyFilters(baseQuery, filters) - baseQuery = baseQuery.Where("status IN ?", []string{"success", "error"}) + baseQuery = baseQuery.Where("status IN ?", terminalLogStatuses) // Query grouped by bucket and model with status counts var results []struct { @@ -1471,35 +1468,17 @@ func (s *RDBLogStore) GetModelHistogram(ctx context.Context, filters SearchFilte Total int64 `gorm:"column:total"` Success int64 `gorm:"column:success"` Error int64 `gorm:"column:error_count"` + Cancelled int64 `gorm:"column:cancelled_count"` } - var selectClause string - switch dialect { - case "sqlite": - selectClause = fmt.Sprintf(` - (CAST(strftime('%%s', timestamp) AS INTEGER) / %d) * %d as bucket_timestamp, - model, - COUNT(*) as total, - SUM(CASE WHEN status = 'success' THEN 1 ELSE 0 END) as success, - SUM(CASE WHEN status = 'error' THEN 1 ELSE 0 END) as error_count - `, bucketSizeSeconds, bucketSizeSeconds) - case "mysql": - selectClause = fmt.Sprintf(` - (FLOOR(UNIX_TIMESTAMP(timestamp) / %d) * %d) as bucket_timestamp, - model, - COUNT(*) as total, - SUM(CASE WHEN status = 'success' THEN 1 ELSE 0 END) as success, - SUM(CASE WHEN status = 'error' THEN 1 ELSE 0 END) as error_count - `, bucketSizeSeconds, bucketSizeSeconds) - default: - selectClause = fmt.Sprintf(` - CAST(FLOOR(EXTRACT(EPOCH FROM timestamp) / %d) * %d AS BIGINT) as bucket_timestamp, + selectClause := fmt.Sprintf(` + %s as bucket_timestamp, model, COUNT(*) as total, SUM(CASE WHEN status = 'success' THEN 1 ELSE 0 END) as success, - SUM(CASE WHEN status = 'error' THEN 1 ELSE 0 END) as error_count - `, bucketSizeSeconds, bucketSizeSeconds) - } + SUM(CASE WHEN status = 'error' THEN 1 ELSE 0 END) as error_count, + SUM(CASE WHEN status = 'cancelled' THEN 1 ELSE 0 END) as cancelled_count + `, unixBucketExpr(dialect, bucketSizeSeconds)) if err := baseQuery. Select(selectClause). @@ -1517,18 +1496,20 @@ func (s *RDBLogStore) GetModelHistogram(ctx context.Context, filters SearchFilte modelsSet[r.Model] = true if bucket, exists := bucketMap[r.BucketTimestamp]; exists { bucket.ByModel[r.Model] = ModelUsageStats{ - Total: r.Total, - Success: r.Success, - Error: r.Error, + Total: r.Total, + Success: r.Success, + Error: r.Error, + Cancelled: r.Cancelled, } } else { bucketMap[r.BucketTimestamp] = &ModelHistogramBucket{ Timestamp: time.Unix(r.BucketTimestamp, 0).UTC(), ByModel: map[string]ModelUsageStats{ r.Model: { - Total: r.Total, - Success: r.Success, - Error: r.Error, + Total: r.Total, + Success: r.Success, + Error: r.Error, + Cancelled: r.Cancelled, }, }, } @@ -1617,7 +1598,7 @@ func (s *RDBLogStore) GetLatencyHistogram(ctx context.Context, filters SearchFil baseQuery := s.ScopedDB(ctx).Model(&Log{}) baseQuery = s.applyFilters(baseQuery, filters) - baseQuery = baseQuery.Where("status IN ?", []string{"success", "error"}) + baseQuery = baseQuery.Where("status IN ?", terminalLogStatuses) baseQuery = baseQuery.Where("latency IS NOT NULL") switch dialect { @@ -1625,11 +1606,60 @@ func (s *RDBLogStore) GetLatencyHistogram(ctx context.Context, filters SearchFil return s.getLatencyHistogramSQLite(ctx, baseQuery, filters, bucketSizeSeconds) case "mysql": return s.getLatencyHistogramMySQL(ctx, baseQuery, filters, bucketSizeSeconds) + case "clickhouse": + return s.getLatencyHistogramClickHouse(ctx, baseQuery, filters, bucketSizeSeconds) default: return s.getLatencyHistogramPercentileCont(ctx, baseQuery, filters, bucketSizeSeconds) } } +// getLatencyHistogramClickHouse computes latency percentiles with ClickHouse's +// quantile() aggregate (ClickHouse does not support percentile_cont ... WITHIN +// GROUP). Shape mirrors getLatencyHistogramPercentileCont. +func (s *RDBLogStore) getLatencyHistogramClickHouse(ctx context.Context, baseQuery *gorm.DB, filters SearchFilters, bucketSizeSeconds int64) (*LatencyHistogramResult, error) { + var results []struct { + BucketTimestamp int64 `gorm:"column:bucket_timestamp"` + AvgLatency sql.NullFloat64 `gorm:"column:avg_latency"` + P90Latency sql.NullFloat64 `gorm:"column:p90_latency"` + P95Latency sql.NullFloat64 `gorm:"column:p95_latency"` + P99Latency sql.NullFloat64 `gorm:"column:p99_latency"` + TotalRequests int64 `gorm:"column:total_requests"` + } + + selectClause := fmt.Sprintf(` + %s as bucket_timestamp, + AVG(latency) as avg_latency, + quantile(0.90)(latency) as p90_latency, + quantile(0.95)(latency) as p95_latency, + quantile(0.99)(latency) as p99_latency, + COUNT(*) as total_requests + `, unixBucketExpr("clickhouse", bucketSizeSeconds)) + + if err := baseQuery. + Select(selectClause). + Group("bucket_timestamp"). + Order("bucket_timestamp ASC"). + Find(&results).Error; err != nil { + return nil, fmt.Errorf("failed to get latency histogram: %w", err) + } + + computedBuckets := make(map[int64]LatencyHistogramBucket, len(results)) + var orderedKeys []int64 + for _, r := range results { + orderedKeys = append(orderedKeys, r.BucketTimestamp) + computedBuckets[r.BucketTimestamp] = LatencyHistogramBucket{ + Timestamp: time.Unix(r.BucketTimestamp, 0).UTC(), + AvgLatency: r.AvgLatency.Float64, + P90Latency: r.P90Latency.Float64, + P95Latency: r.P95Latency.Float64, + P99Latency: r.P99Latency.Float64, + TotalRequests: r.TotalRequests, + } + } + + return s.buildLatencyHistogramResult(computedBuckets, orderedKeys, filters, bucketSizeSeconds) +} + // getLatencyHistogramPercentileCont uses database-level percentile_cont for PostgreSQL. // Returns 1 aggregated row per bucket instead of loading all individual latency values. func (s *RDBLogStore) getLatencyHistogramPercentileCont(ctx context.Context, baseQuery *gorm.DB, filters SearchFilters, bucketSizeSeconds int64) (*LatencyHistogramResult, error) { @@ -1840,7 +1870,7 @@ func (s *RDBLogStore) GetModelRankings(ctx context.Context, filters SearchFilter // Query current period currentQuery := s.ScopedDB(ctx).Model(&Log{}) currentQuery = s.applyFilters(currentQuery, filters) - currentQuery = currentQuery.Where("status IN ?", []string{"success", "error"}) + currentQuery = currentQuery.Where("status IN ?", terminalLogStatuses) currentQuery = currentQuery.Where("model IS NOT NULL AND model != ''") var currentResults []struct { @@ -1879,7 +1909,7 @@ func (s *RDBLogStore) GetModelRankings(ctx context.Context, filters SearchFilter prevQuery := s.ScopedDB(ctx).Model(&Log{}) prevQuery = s.applyFilters(prevQuery, prevFilters) - prevQuery = prevQuery.Where("status IN ?", []string{"success", "error"}) + prevQuery = prevQuery.Where("status IN ?", terminalLogStatuses) prevQuery = prevQuery.Where("model IS NOT NULL AND model != ''") // Only fetch previous-period data for (model, provider) pairs that @@ -1981,7 +2011,7 @@ func (s *RDBLogStore) GetUserRankings(ctx context.Context, filters SearchFilters // Query current period currentQuery := s.ScopedDB(ctx).Model(&Log{}) currentQuery = s.applyFilters(currentQuery, filters) - currentQuery = currentQuery.Where("status IN ?", []string{"success", "error"}) + currentQuery = currentQuery.Where("status IN ?", terminalLogStatuses) currentQuery = currentQuery.Where("user_id IS NOT NULL AND user_id != ''") var currentResults []struct { @@ -2013,7 +2043,7 @@ func (s *RDBLogStore) GetUserRankings(ctx context.Context, filters SearchFilters prevQuery := s.ScopedDB(ctx).Model(&Log{}) prevQuery = s.applyFilters(prevQuery, prevFilters) - prevQuery = prevQuery.Where("status IN ?", []string{"success", "error"}) + prevQuery = prevQuery.Where("status IN ?", terminalLogStatuses) prevQuery = prevQuery.Where("user_id IS NOT NULL AND user_id != ''") if len(currentResults) > 0 { @@ -2125,7 +2155,7 @@ func (s *RDBLogStore) GetDimensionRankings(ctx context.Context, filters SearchFi currentQuery := baseTable(s.ScopedDB(ctx)) currentQuery = s.applyFilters(currentQuery, filters) - currentQuery = currentQuery.Where("status IN ?", []string{"success", "error"}) + currentQuery = currentQuery.Where("status IN ?", terminalLogStatuses) currentQuery = currentQuery.Where(fmt.Sprintf("%s IS NOT NULL AND %s != ''", idCol, idCol)) var currentResults []struct { @@ -2164,7 +2194,7 @@ func (s *RDBLogStore) GetDimensionRankings(ctx context.Context, filters SearchFi if fanoutFrom != "" { requestsCountsQuery := baseTable(s.ScopedDB(ctx)) requestsCountsQuery = s.applyFilters(requestsCountsQuery, filters) - requestsCountsQuery = requestsCountsQuery.Where("status IN ?", []string{"success", "error"}) + requestsCountsQuery = requestsCountsQuery.Where("status IN ?", terminalLogStatuses) requestsCountsQuery = requestsCountsQuery.Where(fmt.Sprintf("%s IS NOT NULL AND %s != ''", idCol, idCol)) if err := requestsCountsQuery. Select("COUNT(DISTINCT id) as actual_requests, COUNT(*) as attributed_requests"). @@ -2185,7 +2215,7 @@ func (s *RDBLogStore) GetDimensionRankings(ctx context.Context, filters SearchFi prevQuery := baseTable(s.ScopedDB(ctx)) prevQuery = s.applyFilters(prevQuery, prevFilters) - prevQuery = prevQuery.Where("status IN ?", []string{"success", "error"}) + prevQuery = prevQuery.Where("status IN ?", terminalLogStatuses) prevQuery = prevQuery.Where(fmt.Sprintf("%s IS NOT NULL AND %s != ''", idCol, idCol)) if len(currentResults) > 0 { @@ -2280,7 +2310,7 @@ func (s *RDBLogStore) GetProviderCostHistogram(ctx context.Context, filters Sear baseQuery := s.ScopedDB(ctx).Model(&Log{}) baseQuery = s.applyFilters(baseQuery, filters) - baseQuery = baseQuery.Where("status IN ?", []string{"success", "error"}) + baseQuery = baseQuery.Where("status IN ?", terminalLogStatuses) baseQuery = baseQuery.Where("cost IS NOT NULL AND cost > 0") var results []struct { @@ -2289,27 +2319,11 @@ func (s *RDBLogStore) GetProviderCostHistogram(ctx context.Context, filters Sear TotalCost float64 `gorm:"column:total_cost"` } - var selectClause string - switch dialect { - case "sqlite": - selectClause = fmt.Sprintf(` - (CAST(strftime('%%s', timestamp) AS INTEGER) / %d) * %d as bucket_timestamp, - provider, - COALESCE(SUM(cost), 0) as total_cost - `, bucketSizeSeconds, bucketSizeSeconds) - case "mysql": - selectClause = fmt.Sprintf(` - (FLOOR(UNIX_TIMESTAMP(timestamp) / %d) * %d) as bucket_timestamp, - provider, - COALESCE(SUM(cost), 0) as total_cost - `, bucketSizeSeconds, bucketSizeSeconds) - default: - selectClause = fmt.Sprintf(` - CAST(FLOOR(EXTRACT(EPOCH FROM timestamp) / %d) * %d AS BIGINT) as bucket_timestamp, + selectClause := fmt.Sprintf(` + %s as bucket_timestamp, provider, COALESCE(SUM(cost), 0) as total_cost - `, bucketSizeSeconds, bucketSizeSeconds) - } + `, unixBucketExpr(dialect, bucketSizeSeconds)) if err := baseQuery. Select(selectClause). @@ -2391,7 +2405,7 @@ func (s *RDBLogStore) GetProviderTokenHistogram(ctx context.Context, filters Sea baseQuery := s.ScopedDB(ctx).Model(&Log{}) baseQuery = s.applyFilters(baseQuery, filters) - baseQuery = baseQuery.Where("status IN ?", []string{"success", "error"}) + baseQuery = baseQuery.Where("status IN ?", terminalLogStatuses) var results []struct { BucketTimestamp int64 `gorm:"column:bucket_timestamp"` @@ -2401,33 +2415,13 @@ func (s *RDBLogStore) GetProviderTokenHistogram(ctx context.Context, filters Sea TotalTokens int64 `gorm:"column:total_tokens"` } - var selectClause string - switch dialect { - case "sqlite": - selectClause = fmt.Sprintf(` - (CAST(strftime('%%s', timestamp) AS INTEGER) / %d) * %d as bucket_timestamp, - provider, - COALESCE(SUM(prompt_tokens), 0) as prompt_tokens, - COALESCE(SUM(completion_tokens), 0) as completion_tokens, - COALESCE(SUM(total_tokens), 0) as total_tokens - `, bucketSizeSeconds, bucketSizeSeconds) - case "mysql": - selectClause = fmt.Sprintf(` - (FLOOR(UNIX_TIMESTAMP(timestamp) / %d) * %d) as bucket_timestamp, - provider, - COALESCE(SUM(prompt_tokens), 0) as prompt_tokens, - COALESCE(SUM(completion_tokens), 0) as completion_tokens, - COALESCE(SUM(total_tokens), 0) as total_tokens - `, bucketSizeSeconds, bucketSizeSeconds) - default: - selectClause = fmt.Sprintf(` - CAST(FLOOR(EXTRACT(EPOCH FROM timestamp) / %d) * %d AS BIGINT) as bucket_timestamp, + selectClause := fmt.Sprintf(` + %s as bucket_timestamp, provider, COALESCE(SUM(prompt_tokens), 0) as prompt_tokens, COALESCE(SUM(completion_tokens), 0) as completion_tokens, COALESCE(SUM(total_tokens), 0) as total_tokens - `, bucketSizeSeconds, bucketSizeSeconds) - } + `, unixBucketExpr(dialect, bucketSizeSeconds)) if err := baseQuery. Select(selectClause). @@ -2518,7 +2512,7 @@ func (s *RDBLogStore) GetProviderLatencyHistogram(ctx context.Context, filters S baseQuery := s.ScopedDB(ctx).Model(&Log{}) baseQuery = s.applyFilters(baseQuery, filters) - baseQuery = baseQuery.Where("status IN ?", []string{"success", "error"}) + baseQuery = baseQuery.Where("status IN ?", terminalLogStatuses) baseQuery = baseQuery.Where("latency IS NOT NULL") switch dialect { @@ -2526,11 +2520,81 @@ func (s *RDBLogStore) GetProviderLatencyHistogram(ctx context.Context, filters S return s.getProviderLatencyHistogramSQLite(ctx, baseQuery, filters, bucketSizeSeconds) case "mysql": return s.getProviderLatencyHistogramMySQL(ctx, baseQuery, filters, bucketSizeSeconds) + case "clickhouse": + return s.getProviderLatencyHistogramClickHouse(ctx, baseQuery, filters, bucketSizeSeconds) default: return s.getProviderLatencyHistogramPercentileCont(ctx, baseQuery, filters, bucketSizeSeconds) } } +// getProviderLatencyHistogramClickHouse computes per-provider latency +// percentiles with ClickHouse's quantile() aggregate. Shape mirrors +// getProviderLatencyHistogramPercentileCont. +func (s *RDBLogStore) getProviderLatencyHistogramClickHouse(ctx context.Context, baseQuery *gorm.DB, filters SearchFilters, bucketSizeSeconds int64) (*ProviderLatencyHistogramResult, error) { + var results []struct { + BucketTimestamp int64 `gorm:"column:bucket_timestamp"` + Provider string `gorm:"column:provider"` + AvgLatency sql.NullFloat64 `gorm:"column:avg_latency"` + P90Latency sql.NullFloat64 `gorm:"column:p90_latency"` + P95Latency sql.NullFloat64 `gorm:"column:p95_latency"` + P99Latency sql.NullFloat64 `gorm:"column:p99_latency"` + TotalRequests int64 `gorm:"column:total_requests"` + } + + selectClause := fmt.Sprintf(` + %s as bucket_timestamp, + provider, + AVG(latency) as avg_latency, + quantile(0.90)(latency) as p90_latency, + quantile(0.95)(latency) as p95_latency, + quantile(0.99)(latency) as p99_latency, + COUNT(*) as total_requests + `, unixBucketExpr("clickhouse", bucketSizeSeconds)) + + if err := baseQuery. + Select(selectClause). + Group("bucket_timestamp, provider"). + Order("bucket_timestamp ASC, provider ASC"). + Find(&results).Error; err != nil { + return nil, fmt.Errorf("failed to get provider latency histogram: %w", err) + } + + providersSet := make(map[string]bool) + computedBuckets := make(map[int64]*ProviderLatencyHistogramBucket) + var orderedBuckets []int64 + seenBuckets := make(map[int64]bool) + + for _, r := range results { + providersSet[r.Provider] = true + if !seenBuckets[r.BucketTimestamp] { + seenBuckets[r.BucketTimestamp] = true + orderedBuckets = append(orderedBuckets, r.BucketTimestamp) + } + stats := ProviderLatencyStats{ + AvgLatency: r.AvgLatency.Float64, + P90Latency: r.P90Latency.Float64, + P95Latency: r.P95Latency.Float64, + P99Latency: r.P99Latency.Float64, + TotalRequests: r.TotalRequests, + } + if bucket, exists := computedBuckets[r.BucketTimestamp]; exists { + bucket.ByProvider[r.Provider] = stats + } else { + computedBuckets[r.BucketTimestamp] = &ProviderLatencyHistogramBucket{ + Timestamp: time.Unix(r.BucketTimestamp, 0).UTC(), + ByProvider: map[string]ProviderLatencyStats{r.Provider: stats}, + } + } + } + + providers := make([]string, 0, len(providersSet)) + for provider := range providersSet { + providers = append(providers, provider) + } + + return s.buildProviderLatencyHistogramResult(computedBuckets, orderedBuckets, providers, filters, bucketSizeSeconds) +} + // getProviderLatencyHistogramPercentileCont uses database-level percentile_cont for PostgreSQL. // Returns 1 aggregated row per (bucket, provider) instead of loading all individual latency values. func (s *RDBLogStore) getProviderLatencyHistogramPercentileCont(ctx context.Context, baseQuery *gorm.DB, filters SearchFilters, bucketSizeSeconds int64) (*ProviderLatencyHistogramResult, error) { @@ -2814,16 +2878,10 @@ func (s *RDBLogStore) GetDimensionCostHistogram(ctx context.Context, filters Sea baseQuery = baseQuery.Model(&Log{}) } baseQuery = s.applyFilters(baseQuery, filters) - baseQuery = baseQuery.Where("status IN ?", []string{"success", "error"}) + baseQuery = baseQuery.Where("status IN ?", terminalLogStatuses) baseQuery = baseQuery.Where("cost IS NOT NULL AND cost > 0") - var bucketExpr string - switch dialect { - case "sqlite": - bucketExpr = fmt.Sprintf("CAST((CAST(strftime('%%s', timestamp) AS INTEGER) / %d) * %d AS INTEGER)", bucketSizeSeconds, bucketSizeSeconds) - default: - bucketExpr = fmt.Sprintf("CAST(FLOOR(EXTRACT(EPOCH FROM timestamp) / %d) * %d AS BIGINT)", bucketSizeSeconds, bucketSizeSeconds) - } + bucketExpr := unixBucketExpr(dialect, bucketSizeSeconds) var results []struct { BucketTimestamp int64 `gorm:"column:bucket_timestamp"` @@ -2923,15 +2981,9 @@ func (s *RDBLogStore) GetDimensionTokenHistogram(ctx context.Context, filters Se baseQuery = baseQuery.Model(&Log{}) } baseQuery = s.applyFilters(baseQuery, filters) - baseQuery = baseQuery.Where("status IN ?", []string{"success", "error"}) + baseQuery = baseQuery.Where("status IN ?", terminalLogStatuses) - var bucketExpr string - switch dialect { - case "sqlite": - bucketExpr = fmt.Sprintf("CAST((CAST(strftime('%%s', timestamp) AS INTEGER) / %d) * %d AS INTEGER)", bucketSizeSeconds, bucketSizeSeconds) - default: - bucketExpr = fmt.Sprintf("CAST(FLOOR(EXTRACT(EPOCH FROM timestamp) / %d) * %d AS BIGINT)", bucketSizeSeconds, bucketSizeSeconds) - } + bucketExpr := unixBucketExpr(dialect, bucketSizeSeconds) var results []struct { BucketTimestamp int64 `gorm:"column:bucket_timestamp"` @@ -3039,16 +3091,10 @@ func (s *RDBLogStore) GetDimensionLatencyHistogram(ctx context.Context, filters dialect := s.db.Dialector.Name() baseQuery := s.ScopedDB(ctx).Model(&Log{}) baseQuery = s.applyFilters(baseQuery, filters) - baseQuery = baseQuery.Where("status IN ?", []string{"success", "error"}) + baseQuery = baseQuery.Where("status IN ?", terminalLogStatuses) baseQuery = baseQuery.Where("latency IS NOT NULL") - var bucketExpr string - switch dialect { - case "sqlite": - bucketExpr = fmt.Sprintf("CAST((CAST(strftime('%%s', timestamp) AS INTEGER) / %d) * %d AS INTEGER)", bucketSizeSeconds, bucketSizeSeconds) - default: - bucketExpr = fmt.Sprintf("CAST(FLOOR(EXTRACT(EPOCH FROM timestamp) / %d) * %d AS BIGINT)", bucketSizeSeconds, bucketSizeSeconds) - } + bucketExpr := unixBucketExpr(dialect, bucketSizeSeconds) var results []struct { BucketTimestamp int64 `gorm:"column:bucket_timestamp"` @@ -3364,9 +3410,12 @@ func (s *RDBLogStore) GetDistinctMetadataKeys(ctx context.Context, limit int, qu var metadataStrings []string // Guard must match the partial-index predicate so the planner uses the GIN index. var metadataGuard string - if s.db.Dialector.Name() == "postgres" { + switch s.db.Dialector.Name() { + case "postgres": metadataGuard = "metadata IS NOT NULL AND metadata IS JSON OBJECT AND metadata != '{}' AND timestamp >= ?" - } else { + case "clickhouse": + metadataGuard = "metadata IS NOT NULL AND isValidJSON(metadata) AND metadata != '{}' AND timestamp >= ?" + default: metadataGuard = "metadata IS NOT NULL AND json_valid(metadata) AND json_type(metadata) = 'object' AND metadata != '{}' AND timestamp >= ?" } err := s.ScopedDB(ctx).Model(&Log{}). @@ -3916,30 +3965,12 @@ func (s *RDBLogStore) GetMCPHistogram(ctx context.Context, filters MCPToolLogSea Error int64 `gorm:"column:error"` } - var selectClause string - switch dialect { - case "sqlite": - selectClause = fmt.Sprintf(` - (CAST(strftime('%%s', timestamp) AS INTEGER) / %d) * %d as bucket_timestamp, - COUNT(*) as count, - SUM(CASE WHEN status = 'success' THEN 1 ELSE 0 END) as success, - SUM(CASE WHEN status = 'error' THEN 1 ELSE 0 END) as error - `, bucketSizeSeconds, bucketSizeSeconds) - case "mysql": - selectClause = fmt.Sprintf(` - (FLOOR(UNIX_TIMESTAMP(timestamp) / %d) * %d) as bucket_timestamp, - COUNT(*) as count, - SUM(CASE WHEN status = 'success' THEN 1 ELSE 0 END) as success, - SUM(CASE WHEN status = 'error' THEN 1 ELSE 0 END) as error - `, bucketSizeSeconds, bucketSizeSeconds) - default: - selectClause = fmt.Sprintf(` - CAST(FLOOR(EXTRACT(EPOCH FROM timestamp) / %d) * %d AS BIGINT) as bucket_timestamp, + selectClause := fmt.Sprintf(` + %s as bucket_timestamp, COUNT(*) as count, SUM(CASE WHEN status = 'success' THEN 1 ELSE 0 END) as success, SUM(CASE WHEN status = 'error' THEN 1 ELSE 0 END) as error - `, bucketSizeSeconds, bucketSizeSeconds) - } + `, unixBucketExpr(dialect, bucketSizeSeconds)) if err := baseQuery. Select(selectClause). @@ -4011,24 +4042,10 @@ func (s *RDBLogStore) GetMCPCostHistogram(ctx context.Context, filters MCPToolLo TotalCost float64 `gorm:"column:total_cost"` } - var selectClause string - switch dialect { - case "sqlite": - selectClause = fmt.Sprintf(` - (CAST(strftime('%%s', timestamp) AS INTEGER) / %d) * %d as bucket_timestamp, - COALESCE(SUM(cost), 0) as total_cost - `, bucketSizeSeconds, bucketSizeSeconds) - case "mysql": - selectClause = fmt.Sprintf(` - (FLOOR(UNIX_TIMESTAMP(timestamp) / %d) * %d) as bucket_timestamp, - COALESCE(SUM(cost), 0) as total_cost - `, bucketSizeSeconds, bucketSizeSeconds) - default: - selectClause = fmt.Sprintf(` - CAST(FLOOR(EXTRACT(EPOCH FROM timestamp) / %d) * %d AS BIGINT) as bucket_timestamp, + selectClause := fmt.Sprintf(` + %s as bucket_timestamp, COALESCE(SUM(cost), 0) as total_cost - `, bucketSizeSeconds, bucketSizeSeconds) - } + `, unixBucketExpr(dialect, bucketSizeSeconds)) if err := baseQuery. Select(selectClause). diff --git a/framework/logstore/rdb_perf_test.go b/framework/logstore/rdb_perf_test.go index 1fb78f00b9d..c54ea67ab1c 100644 --- a/framework/logstore/rdb_perf_test.go +++ b/framework/logstore/rdb_perf_test.go @@ -35,6 +35,125 @@ func newTestSQLiteStore(t *testing.T) *RDBLogStore { return store } +func TestCancelledStatusIncludedInLogAggregates(t *testing.T) { + store := newTestSQLiteStore(t) + ctx := context.Background() + base := time.Date(2026, 1, 2, 3, 4, 0, 0, time.UTC) + cost := 0.01 + successLatency := 100.0 + errorLatency := 200.0 + cancelledLatency := 300.0 + processingLatency := 400.0 + + entries := []*Log{ + { + ID: "aggregate-success", + Timestamp: base, + Object: "chat.completion", + Provider: "openai", + Model: "gpt-4o-mini", + Status: "success", + Latency: &successLatency, + TotalTokens: 10, + Cost: &cost, + }, + { + ID: "aggregate-error", + Timestamp: base.Add(time.Minute), + Object: "chat.completion", + Provider: "openai", + Model: "gpt-4o-mini", + Status: "error", + Latency: &errorLatency, + TotalTokens: 20, + Cost: &cost, + }, + { + ID: "aggregate-cancelled", + Timestamp: base.Add(2 * time.Minute), + Object: "chat.completion", + Provider: "openai", + Model: "gpt-4o-mini", + Status: "cancelled", + Latency: &cancelledLatency, + TotalTokens: 30, + Cost: &cost, + }, + { + ID: "aggregate-processing", + Timestamp: base.Add(3 * time.Minute), + Object: "chat.completion", + Provider: "openai", + Model: "gpt-4o-mini", + Status: "processing", + Latency: &processingLatency, + }, + } + + for _, entry := range entries { + if err := store.Create(ctx, entry); err != nil { + t.Fatalf("Create(%s) error = %v", entry.ID, err) + } + } + + search, err := store.SearchLogs(ctx, SearchFilters{Status: []string{"cancelled"}}, PaginationOptions{Limit: 10}) + if err != nil { + t.Fatalf("SearchLogs(cancelled) error = %v", err) + } + if search.Stats.TotalRequests != 1 { + t.Fatalf("expected cancelled search total to be 1, got %d", search.Stats.TotalRequests) + } + + allStats, err := store.GetStats(ctx, SearchFilters{}) + if err != nil { + t.Fatalf("GetStats(all) error = %v", err) + } + if allStats.TotalRequests != 4 { + t.Fatalf("expected total requests to include processing rows, got %d", allStats.TotalRequests) + } + if allStats.CacheHitRateTotalRequests == nil || *allStats.CacheHitRateTotalRequests != 3 { + t.Fatalf("expected terminal request denominator to include cancelled rows, got %v", allStats.CacheHitRateTotalRequests) + } + + cancelledStats, err := store.GetStats(ctx, SearchFilters{Status: []string{"cancelled"}}) + if err != nil { + t.Fatalf("GetStats(cancelled) error = %v", err) + } + if cancelledStats.TotalRequests != 1 { + t.Fatalf("expected cancelled stats total to be 1, got %d", cancelledStats.TotalRequests) + } + if cancelledStats.SuccessRate != 0 { + t.Fatalf("expected cancelled success rate to be 0, got %f", cancelledStats.SuccessRate) + } + if cancelledStats.AverageLatency != 300 { + t.Fatalf("expected cancelled average latency 300, got %f", cancelledStats.AverageLatency) + } + + start := base.Add(-time.Minute) + end := base.Add(5 * time.Minute) + hist, err := store.GetHistogram(ctx, SearchFilters{ + Status: []string{"cancelled"}, + StartTime: &start, + EndTime: &end, + }, 3600) + if err != nil { + t.Fatalf("GetHistogram(cancelled) error = %v", err) + } + var cancelledBucket *HistogramBucket + for i := range hist.Buckets { + if hist.Buckets[i].Count > 0 { + cancelledBucket = &hist.Buckets[i] + break + } + } + if cancelledBucket == nil { + t.Fatal("expected a non-empty cancelled histogram bucket") + } + if cancelledBucket.Count != 1 || cancelledBucket.Success != 0 || cancelledBucket.Error != 0 || cancelledBucket.Cancelled != 1 { + t.Fatalf("unexpected cancelled histogram bucket: %+v", *cancelledBucket) + } +} + func TestLogCreateSerializesFields(t *testing.T) { store := newTestSQLiteStore(t) prompt := "hello" diff --git a/framework/logstore/store.go b/framework/logstore/store.go index 56e8be8b3d4..3c59d580304 100644 --- a/framework/logstore/store.go +++ b/framework/logstore/store.go @@ -14,8 +14,9 @@ type LogStoreType string // LogStoreTypeSQLite is the type of log store for SQLite. const ( - LogStoreTypeSQLite LogStoreType = "sqlite" - LogStoreTypePostgres LogStoreType = "postgres" + LogStoreTypeSQLite LogStoreType = "sqlite" + LogStoreTypePostgres LogStoreType = "postgres" + LogStoreTypeClickHouse LogStoreType = "clickhouse" ) // LogStore is the interface for the log store. @@ -122,6 +123,12 @@ func NewLogStore(ctx context.Context, config *Config, logger schemas.Logger) (Lo } else { return nil, fmt.Errorf("invalid postgres config: %T", config.Config) } + case LogStoreTypeClickHouse: + if clickhouseConfig, ok := config.Config.(*ClickHouseConfig); ok { + inner, err = newClickHouseLogStore(ctx, clickhouseConfig, config.RetentionDays, logger) + } else { + return nil, fmt.Errorf("invalid clickhouse config: %T", config.Config) + } default: return nil, fmt.Errorf("unsupported log store type: %s", config.Type) } diff --git a/framework/logstore/tables.go b/framework/logstore/tables.go index ad0122f8a15..61c2d263852 100644 --- a/framework/logstore/tables.go +++ b/framework/logstore/tables.go @@ -1349,6 +1349,7 @@ type HistogramBucket struct { Count int64 `json:"count"` Success int64 `json:"success"` Error int64 `json:"error"` + Cancelled int64 `json:"cancelled"` } // HistogramResult represents the histogram query result @@ -1388,9 +1389,10 @@ type CostHistogramResult struct { // ModelUsageStats represents usage statistics for a single model type ModelUsageStats struct { - Total int64 `json:"total"` - Success int64 `json:"success"` - Error int64 `json:"error"` + Total int64 `json:"total"` + Success int64 `json:"success"` + Error int64 `json:"error"` + Cancelled int64 `json:"cancelled"` } // ModelHistogramBucket represents a single time bucket for model usage diff --git a/framework/mcp_headers/sweep.go b/framework/mcp_headers/sweep.go index 4d50c68d209..a198d335fc4 100644 --- a/framework/mcp_headers/sweep.go +++ b/framework/mcp_headers/sweep.go @@ -32,6 +32,7 @@ type CredentialSweepWorker struct { expiredFlowEvery time.Duration stopCh chan struct{} stopOnce sync.Once + cancel context.CancelFunc logger schemas.Logger } @@ -57,7 +58,9 @@ func NewCredentialSweepWorker(provider *Provider, orphanRetention time.Duration, // Start begins the sweep worker in a background goroutine. func (w *CredentialSweepWorker) Start(ctx context.Context) { - go w.run(ctx) + runCtx, cancel := context.WithCancel(ctx) + w.cancel = cancel + go w.run(runCtx) if w.logger != nil { w.logger.Info("Per-user headers sweep worker started (orphan=%s, retention=%s, expired_flow=%s)", w.orphanSweepEvery, w.orphanRetention, w.expiredFlowEvery) @@ -68,6 +71,11 @@ func (w *CredentialSweepWorker) Start(ctx context.Context) { // panics from redundant shutdown paths. func (w *CredentialSweepWorker) Stop() { w.stopOnce.Do(func() { + // Cancel any in-flight sweep so a blocked DB call unwinds promptly, + // then signal run() to exit its ticker loop. + if w.cancel != nil { + w.cancel() + } close(w.stopCh) if w.logger != nil { w.logger.Info("Per-user headers credential sweep worker stopped") diff --git a/framework/modelcatalog/datasheet/capabilities_test.go b/framework/modelcatalog/datasheet/capabilities_test.go index 555ec46ecfc..1e337d4857c 100644 --- a/framework/modelcatalog/datasheet/capabilities_test.go +++ b/framework/modelcatalog/datasheet/capabilities_test.go @@ -189,9 +189,10 @@ func TestCapabilityFieldsRoundTripThroughPricingConversions(t *testing.T) { inputCost := float64(1) outputCost := float64(2) entry := Entry{ - BaseModel: "gpt-4o", - Provider: "openai", - Mode: "chat", + BaseModel: "gpt-4o", + Provider: "openai", + Mode: "chat", + IsDeprecated: true, Options: Options{ InputCostPerToken: &inputCost, OutputCostPerToken: &outputCost, @@ -219,6 +220,9 @@ func TestCapabilityFieldsRoundTripThroughPricingConversions(t *testing.T) { if roundTrip.Architecture == nil || roundTrip.Architecture.Modality == nil || *roundTrip.Architecture.Modality != modality { t.Fatalf("expected architecture to round-trip, got %#v", roundTrip.Architecture) } + if !roundTrip.IsDeprecated { + t.Fatalf("expected is_deprecated to round-trip") + } } func capabilityIntPtr(v int) *int { return &v } diff --git a/framework/modelcatalog/datasheet/cost.go b/framework/modelcatalog/datasheet/cost.go index 3091a09ed3f..c7a89b2388d 100644 --- a/framework/modelcatalog/datasheet/cost.go +++ b/framework/modelcatalog/datasheet/cost.go @@ -386,7 +386,6 @@ func computeTextCost(pricing *configstoreTables.TableModelPricing, usage *schema return 0 } - totalTokens := usage.TotalTokens promptTokens := usage.PromptTokens completionTokens := usage.CompletionTokens @@ -402,11 +401,15 @@ func computeTextCost(pricing *configstoreTables.TableModelPricing, usage *schema } } - inputRate := tieredInputRate(pricing, totalTokens, tier) - outputRate := tieredOutputRate(pricing, totalTokens, tier) - cacheReadInputRate := tieredCacheReadInputTokenRate(pricing, totalTokens, tier) - cacheCreationInputRate := tieredCacheCreationInputTokenRate(pricing, totalTokens, tier) - cacheCreationInputAbove1hrInputRate := tieredCacheCreationInputAbove1hrTokenRate(pricing, totalTokens, tier) + // Long-context pricing tiers are selected by input context size. Once + // selected, the tier's input/cache/output rates apply to their respective + // billed token categories for the request. + tierTokens := promptTokens + inputRate := tieredInputRate(pricing, tierTokens, tier) + outputRate := tieredOutputRate(pricing, tierTokens, tier) + cacheReadInputRate := tieredCacheReadInputTokenRate(pricing, tierTokens, tier) + cacheCreationInputRate := tieredCacheCreationInputTokenRate(pricing, tierTokens, tier) + cacheCreationInputAbove1hrInputRate := tieredCacheCreationInputAbove1hrTokenRate(pricing, tierTokens, tier) // Clamp cached token counts to avoid negative billing on malformed provider payloads if cachedReadTokens > promptTokens { @@ -483,7 +486,7 @@ func computeEmbeddingCost(pricing *configstoreTables.TableModelPricing, usage *s if usage == nil { return 0 } - return float64(usage.PromptTokens) * tieredInputRate(pricing, usage.TotalTokens, tier) + return float64(usage.PromptTokens) * tieredInputRate(pricing, usage.PromptTokens, tier) } // computeRerankCost handles rerank requests. @@ -491,8 +494,9 @@ func computeRerankCost(pricing *configstoreTables.TableModelPricing, usage *sche if usage == nil { return 0 } - inputCost := float64(usage.PromptTokens) * tieredInputRate(pricing, usage.TotalTokens, tier) - outputCost := float64(usage.CompletionTokens) * tieredOutputRate(pricing, usage.TotalTokens, tier) + tierTokens := usage.PromptTokens + inputCost := float64(usage.PromptTokens) * tieredInputRate(pricing, tierTokens, tier) + outputCost := float64(usage.CompletionTokens) * tieredOutputRate(pricing, tierTokens, tier) searchCost := 0.0 if pricing.SearchContextCostPerQuery != nil && usage.CompletionTokensDetails != nil && usage.CompletionTokensDetails.NumSearchQueries != nil { @@ -511,7 +515,7 @@ func computeRerankCost(pricing *configstoreTables.TableModelPricing, usage *sche // since TTS providers report their billable unit in that field. // Output falls back to per-second duration when no audio token rate is configured. func computeSpeechCost(pricing *configstoreTables.TableModelPricing, usage *schemas.BifrostLLMUsage, audioSeconds *int, audioTextInputChars int, tier serviceTier) float64 { - totalTokens := safeTotalTokens(usage) + tierTokens := inputTierTokens(usage) // Input: per-character rate takes precedence for TTS/audio models inputCost := 0.0 @@ -519,14 +523,14 @@ func computeSpeechCost(pricing *configstoreTables.TableModelPricing, usage *sche if pricing.InputCostPerCharacter != nil { inputCost = float64(audioTextInputChars) * *pricing.InputCostPerCharacter } else { - inputCost = float64(audioTextInputChars) * tieredInputRate(pricing, totalTokens, tier) + inputCost = float64(audioTextInputChars) * tieredInputRate(pricing, tierTokens, tier) } } else if usage != nil && usage.PromptTokens > 0 { - inputCost = float64(usage.PromptTokens) * tieredInputRate(pricing, totalTokens, tier) + inputCost = float64(usage.PromptTokens) * tieredInputRate(pricing, tierTokens, tier) } // Output: audio tokens first, then per-second fallback - outputCost := computeAudioOutputCost(pricing, usage, audioSeconds, totalTokens, tier) + outputCost := computeAudioOutputCost(pricing, usage, audioSeconds, tierTokens, tier) return inputCost + outputCost } @@ -535,15 +539,15 @@ func computeSpeechCost(pricing *configstoreTables.TableModelPricing, usage *sche // Input is audio, output is text (CompletionTokens). // Input and output are calculated independently — tokens first, then per-second fallback. func computeTranscriptionCost(pricing *configstoreTables.TableModelPricing, usage *schemas.BifrostLLMUsage, audioSeconds *int, audioTokenDetails *schemas.TranscriptionUsageInputTokenDetails, tier serviceTier) float64 { - totalTokens := safeTotalTokens(usage) + tierTokens := inputTierTokens(usage) // Input: audio tokens/details first, then per-second fallback - inputCost := computeAudioInputCost(pricing, usage, audioSeconds, audioTokenDetails, totalTokens, tier) + inputCost := computeAudioInputCost(pricing, usage, audioSeconds, audioTokenDetails, tierTokens, tier) // Output: text tokens outputCost := 0.0 if usage != nil && usage.CompletionTokens > 0 { - outputCost = float64(usage.CompletionTokens) * tieredOutputRate(pricing, totalTokens, tier) + outputCost = float64(usage.CompletionTokens) * tieredOutputRate(pricing, tierTokens, tier) } return inputCost + outputCost @@ -600,10 +604,10 @@ func computeImageCost(pricing *configstoreTables.TableModelPricing, imageUsage * return 0 } - totalTokens := imageUsage.TotalTokens + tierTokens := imageInputTierTokens(imageUsage) pixels := parseImagePixels(imageSize) - inputCost := computeImageInputCost(pricing, imageUsage, totalTokens, pixels, tier) - outputCost := computeImageOutputCost(pricing, imageUsage, totalTokens, pixels, imageQuality, tier) + inputCost := computeImageInputCost(pricing, imageUsage, tierTokens, pixels, tier) + outputCost := computeImageOutputCost(pricing, imageUsage, tierTokens, pixels, imageQuality, tier) return inputCost + outputCost } @@ -720,14 +724,14 @@ func computeImageOutputCost(pricing *configstoreTables.TableModelPricing, imageU // computeVideoCost handles video generation requests. // Input and output are calculated independently — tokens first, then per-second fallback. func computeVideoCost(pricing *configstoreTables.TableModelPricing, usage *schemas.BifrostLLMUsage, videoSeconds *int, tier serviceTier) float64 { - totalTokens := safeTotalTokens(usage) + tierTokens := inputTierTokens(usage) // Input: text prompt tokens first, then per-second fallback inputCost := 0.0 if usage != nil && usage.PromptTokens > 0 { - inputCost = float64(usage.PromptTokens) * tieredInputRate(pricing, totalTokens, tier) + inputCost = float64(usage.PromptTokens) * tieredInputRate(pricing, tierTokens, tier) } else if videoSeconds != nil && *videoSeconds > 0 { - if rate := tieredVideoInputPerSecondRate(pricing, totalTokens); rate > 0 { + if rate := tieredVideoInputPerSecondRate(pricing, tierTokens); rate > 0 { inputCost = float64(*videoSeconds) * rate } } @@ -735,7 +739,7 @@ func computeVideoCost(pricing *configstoreTables.TableModelPricing, usage *schem // Output: completion tokens first, then per-second fallback outputCost := 0.0 if usage != nil && usage.CompletionTokens > 0 { - outputCost = float64(usage.CompletionTokens) * tieredOutputRate(pricing, totalTokens, tier) + outputCost = float64(usage.CompletionTokens) * tieredOutputRate(pricing, tierTokens, tier) } else if videoSeconds != nil && *videoSeconds > 0 { if pricing.OutputCostPerVideoPerSecond != nil { outputCost = float64(*videoSeconds) * *pricing.OutputCostPerVideoPerSecond @@ -984,11 +988,42 @@ func tieredCacheCreationInputAbove1hrTokenRate(pricing *configstoreTables.TableM return tieredCacheCreationInputTokenRate(pricing, totalTokens, tier) } -func safeTotalTokens(usage *schemas.BifrostLLMUsage) int { +func inputTierTokens(usage *schemas.BifrostLLMUsage) int { if usage == nil { return 0 } - return usage.TotalTokens + return usage.PromptTokens +} + +func imageInputTierTokens(usage *schemas.ImageUsage) int { + if usage == nil { + return 0 + } + if usage.InputTokensDetails != nil { + return usage.InputTokensDetails.TextTokens + usage.InputTokensDetails.ImageTokens + } + if usage.InputTokens > 0 { + return usage.InputTokens + } + + // Some older/provider-specific image adapters only report TotalTokens. + // Derive input from total-output when output is known, but do not treat a + // bare total as input: total includes generated output tokens. + outputTokens := imageOutputTokens(usage) + if usage.TotalTokens > outputTokens { + return usage.TotalTokens - outputTokens + } + return 0 +} + +func imageOutputTokens(usage *schemas.ImageUsage) int { + if usage == nil { + return 0 + } + if usage.OutputTokensDetails != nil { + return usage.OutputTokensDetails.TextTokens + usage.OutputTokensDetails.ImageTokens + } + return usage.OutputTokens } // parseImagePixels parses a size string like "1024x1024" into total pixel count. @@ -1096,7 +1131,8 @@ func (s *Store) resolvePricing(routingInfo schemas.RoutingInfo, requestType sche // // - Gemini: retries under the "vertex" provider, then falls back to chat mode for Responses requests. // - Vertex: strips the "provider/model" prefix and retries, then falls back to chat mode for Responses requests. -// - Bedrock: prepends the "anthropic." namespace for Claude models, then falls back to chat mode for Responses requests. +// - Bedrock: prepends the vendor namespace ("anthropic.", "openai.", "google.", "xai.") inferred from the model family, then falls back to chat mode for Responses requests. +// - Bedrock Mantle: folded onto the "bedrock" provider up front (datasheet rows for all Bedrock variants are stored there), so it shares every Bedrock fallback. // - All providers: for Responses/ResponsesStream requests, retries the lookup in chat mode. // - All providers: for ImageEdit/ImageVariation requests, retries the lookup in image-generation mode. // @@ -1116,6 +1152,13 @@ func (s *Store) getBasePricing(model, provider string, requestType schemas.Reque mode := normalizeRequestType(requestType) + // Datasheet rows for all Bedrock variants are stored under the "bedrock" + // provider (normalizeProvider folds bedrock_* onto "bedrock"), so + // bedrock_mantle lookups run entirely against "bedrock". + if provider == string(schemas.BedrockMantle) { + provider = string(schemas.Bedrock) + } + pricing, ok := s.pricingData[makeKey(model, provider, mode)] if ok { return &pricing, true @@ -1161,10 +1204,24 @@ func (s *Store) getBasePricing(model, provider string, requestType schemas.Reque } if provider == string(schemas.Bedrock) { - // If model is claude without "anthropic." prefix, try with "anthropic." prefix - if !strings.Contains(model, "anthropic.") && schemas.IsAnthropicModel(model) { - s.logger.Debug("primary lookup failed, trying with anthropic. prefix for the same model") - pricing, ok = s.pricingData[makeKey("anthropic."+model, provider, mode)] + // Bedrock model IDs carry a vendor namespace ("anthropic.claude-*", + // "openai.gpt-oss-*", "google.gemma-*", "xai.grok-*"). When the caller + // sends the bare model name, retry with the namespace inferred from + // the model family. + var vendorPrefix string + switch { + case !strings.Contains(model, "anthropic.") && schemas.IsAnthropicModel(model): + vendorPrefix = "anthropic." + case !strings.Contains(model, "openai.") && schemas.IsOpenAIModel(model): + vendorPrefix = "openai." + case !strings.Contains(model, "google.") && (schemas.IsGemmaModel(model) || schemas.IsGeminiModel(model)): + vendorPrefix = "google." + case !strings.Contains(model, "xai.") && schemas.IsGrokModel(model): + vendorPrefix = "xai." + } + if vendorPrefix != "" { + s.logger.Debug("primary lookup failed, trying with %s prefix for the same model", vendorPrefix) + pricing, ok = s.pricingData[makeKey(vendorPrefix+model, provider, mode)] if ok { return &pricing, true } @@ -1172,7 +1229,7 @@ func (s *Store) getBasePricing(model, provider string, requestType schemas.Reque // Lookup in chat if responses not found if requestType == schemas.ResponsesRequest || requestType == schemas.ResponsesStreamRequest || requestType == schemas.WebSocketResponsesRequest || requestType == schemas.RealtimeRequest || requestType == schemas.CompactionRequest { s.logger.Debug("secondary lookup failed, trying chat provider for the same model in chat completion") - pricing, ok = s.pricingData[makeKey("anthropic."+model, provider, normalizeRequestType(schemas.ChatCompletionRequest))] + pricing, ok = s.pricingData[makeKey(vendorPrefix+model, provider, normalizeRequestType(schemas.ChatCompletionRequest))] if ok { return &pricing, true } diff --git a/framework/modelcatalog/datasheet/cost_test.go b/framework/modelcatalog/datasheet/cost_test.go index 6b7cf88d697..7c40b590d06 100644 --- a/framework/modelcatalog/datasheet/cost_test.go +++ b/framework/modelcatalog/datasheet/cost_test.go @@ -395,9 +395,9 @@ func TestComputeTextCost_1hrCacheCreationAbove200k_UsesAbove1hrAbove200kRate(t * p.CacheCreationInputTokenCostAbove1hrAbove200kTokens = bifrost.Ptr(0.000015) usage := &schemas.BifrostLLMUsage{ - PromptTokens: 180000, + PromptTokens: 210000, CompletionTokens: 25000, - TotalTokens: 205000, // above 200k threshold + TotalTokens: 235000, // input is above 200k threshold PromptTokensDetails: &schemas.ChatPromptTokensDetails{ CachedWriteTokens: 10000, CachedWriteTokenDetails: &schemas.ChatCachedWriteTokenDetails{ @@ -409,11 +409,11 @@ func TestComputeTextCost_1hrCacheCreationAbove200k_UsesAbove1hrAbove200kRate(t * cost := computeTextCost(&p, usage, serviceTier{}) // Input rate (>200k): 0.000006; output rate (>200k): 0.00003 - // Input (non-cached): (180000-10000)*0.000006 = 170000*0.000006 = 1.02 + // Input (non-cached): (210000-10000)*0.000006 = 200000*0.000006 = 1.20 // Cache creation 1hr above 200k: 10000*0.000015 = 0.15 // Output: 25000*0.00003 = 0.75 - // Total: 1.02 + 0.15 + 0.75 = 1.92 - assert.InDelta(t, 1.92, cost, 1e-9) + // Total: 1.20 + 0.15 + 0.75 = 2.10 + assert.InDelta(t, 2.10, cost, 1e-9) } func TestComputeTextCost_1hrCacheCreationAbove200k_FallsBackToAbove1hrWhenAbove200kRateAbsent(t *testing.T) { @@ -429,9 +429,9 @@ func TestComputeTextCost_1hrCacheCreationAbove200k_FallsBackToAbove1hrWhenAbove2 // CacheCreationInputTokenCostAbove1hrAbove200kTokens intentionally left nil usage := &schemas.BifrostLLMUsage{ - PromptTokens: 180000, + PromptTokens: 210000, CompletionTokens: 25000, - TotalTokens: 205000, + TotalTokens: 235000, PromptTokensDetails: &schemas.ChatPromptTokensDetails{ CachedWriteTokens: 10000, CachedWriteTokenDetails: &schemas.ChatCachedWriteTokenDetails{ @@ -443,10 +443,10 @@ func TestComputeTextCost_1hrCacheCreationAbove200k_FallsBackToAbove1hrWhenAbove2 cost := computeTextCost(&p, usage, serviceTier{}) // Cache creation 1hr (no above_200k_1hr rate, uses above_1hr): 10000*0.000009 = 0.09 - // Input (non-cached): 170000*0.000006 = 1.02 + // Input (non-cached): 200000*0.000006 = 1.20 // Output: 25000*0.00003 = 0.75 - // Total: 1.02 + 0.09 + 0.75 = 1.86 - assert.InDelta(t, 1.86, cost, 1e-9) + // Total: 1.20 + 0.09 + 0.75 = 2.04 + assert.InDelta(t, 2.04, cost, 1e-9) } func TestComputeTextCost_1hrCacheCreationAbove200k_FallsBackToStandardAbove200kWhenNo1hrRates(t *testing.T) { @@ -460,9 +460,9 @@ func TestComputeTextCost_1hrCacheCreationAbove200k_FallsBackToStandardAbove200kW // Neither CacheCreationInputTokenCostAbove1hr nor Above1hrAbove200k is set usage := &schemas.BifrostLLMUsage{ - PromptTokens: 180000, + PromptTokens: 210000, CompletionTokens: 25000, - TotalTokens: 205000, + TotalTokens: 235000, PromptTokensDetails: &schemas.ChatPromptTokensDetails{ CachedWriteTokens: 10000, CachedWriteTokenDetails: &schemas.ChatCachedWriteTokenDetails{ @@ -474,10 +474,10 @@ func TestComputeTextCost_1hrCacheCreationAbove200k_FallsBackToStandardAbove200kW cost := computeTextCost(&p, usage, serviceTier{}) // Cache creation (1hr → no 1hr rates → standard above_200k): 10000*0.000002 = 0.02 - // Input (non-cached): 170000*0.0000016 = 0.272 + // Input (non-cached): 200000*0.0000016 = 0.32 // Output: 25000*0.000008 = 0.2 - // Total: 0.272 + 0.02 + 0.2 = 0.492 - assert.InDelta(t, 0.492, cost, 1e-9) + // Total: 0.32 + 0.02 + 0.2 = 0.54 + assert.InDelta(t, 0.54, cost, 1e-9) } func TestComputeTextCost_Tiered200k(t *testing.T) { @@ -487,16 +487,16 @@ func TestComputeTextCost_Tiered200k(t *testing.T) { p.OutputCostPerTokenAbove200kTokens = bifrost.Ptr(0.00003) usage := &schemas.BifrostLLMUsage{ - PromptTokens: 180000, + PromptTokens: 210000, CompletionTokens: 30000, - TotalTokens: 210000, // Above 200k threshold + TotalTokens: 240000, // input is above 200k threshold } cost := computeTextCost(&p, usage, serviceTier{}) - // Uses tiered rate since total > 200k - // 180000 * 0.000006 + 30000 * 0.00003 = 1.08 + 0.90 = 1.98 - assert.InDelta(t, 1.98, cost, 1e-9) + // Uses tiered rate since input > 200k + // 210000 * 0.000006 + 30000 * 0.00003 = 1.26 + 0.90 = 2.16 + assert.InDelta(t, 2.16, cost, 1e-9) } func TestComputeTextCost_Below200kUsesBaseRate(t *testing.T) { @@ -512,11 +512,29 @@ func TestComputeTextCost_Below200kUsesBaseRate(t *testing.T) { cost := computeTextCost(&p, usage, serviceTier{}) - // Uses base rate since total < 200k + // Uses base rate since input < 200k // 1000 * 0.000003 + 500 * 0.000015 = 0.003 + 0.0075 = 0.0105 assert.InDelta(t, 0.0105, cost, 1e-12) } +func TestComputeTextCost_TotalAbove200kButInputBelow200kUsesBaseRate(t *testing.T) { + p := chatPricing(0.000003, 0.000015) + p.InputCostPerTokenAbove200kTokens = bifrost.Ptr(0.000006) + p.OutputCostPerTokenAbove200kTokens = bifrost.Ptr(0.00003) + + usage := &schemas.BifrostLLMUsage{ + PromptTokens: 180000, + CompletionTokens: 30000, + TotalTokens: 210000, // total is above 200k, input is not + } + + cost := computeTextCost(&p, usage, serviceTier{}) + + // Uses base rates because long-context tiers are selected by input tokens. + // 180000 * 0.000003 + 30000 * 0.000015 = 0.54 + 0.45 = 0.99 + assert.InDelta(t, 0.99, cost, 1e-9) +} + func TestComputeTextCost_Tiered272k(t *testing.T) { p := chatPricing(0.000003, 0.000015) p.InputCostPerTokenAbove200kTokens = new(0.000006) @@ -525,16 +543,16 @@ func TestComputeTextCost_Tiered272k(t *testing.T) { p.OutputCostPerTokenAbove272kTokens = new(0.000045) usage := &schemas.BifrostLLMUsage{ - PromptTokens: 250000, + PromptTokens: 280000, CompletionTokens: 30000, - TotalTokens: 280000, // Above 272k threshold + TotalTokens: 310000, // input is above 272k threshold } cost := computeTextCost(&p, usage, serviceTier{}) - // Uses 272k tiered rate since total > 272k - // 250000 * 0.000009 + 30000 * 0.000045 = 2.25 + 1.35 = 3.60 - assert.InDelta(t, 3.60, cost, 1e-9) + // Uses 272k tiered rate since input > 272k + // 280000 * 0.000009 + 30000 * 0.000045 = 2.52 + 1.35 = 3.87 + assert.InDelta(t, 3.87, cost, 1e-9) } func TestComputeTextCost_Between200kAnd272kUses200kRate(t *testing.T) { @@ -545,16 +563,16 @@ func TestComputeTextCost_Between200kAnd272kUses200kRate(t *testing.T) { p.OutputCostPerTokenAbove272kTokens = new(0.000045) usage := &schemas.BifrostLLMUsage{ - PromptTokens: 200000, + PromptTokens: 230000, CompletionTokens: 30000, - TotalTokens: 230000, // Between 200k and 272k + TotalTokens: 260000, // input is between 200k and 272k } cost := computeTextCost(&p, usage, serviceTier{}) - // Uses 200k tiered rate since total > 200k but <= 272k - // 200000 * 0.000006 + 30000 * 0.00003 = 1.20 + 0.90 = 2.10 - assert.InDelta(t, 2.10, cost, 1e-9) + // Uses 200k tiered rate since input > 200k but <= 272k + // 230000 * 0.000006 + 30000 * 0.00003 = 1.38 + 0.90 = 2.28 + assert.InDelta(t, 2.28, cost, 1e-9) } func TestComputeTextCost_272kTierWithCacheRead(t *testing.T) { @@ -565,9 +583,9 @@ func TestComputeTextCost_272kTierWithCacheRead(t *testing.T) { p.CacheReadInputTokenCostAbove272kTokens = new(0.0000009) usage := &schemas.BifrostLLMUsage{ - PromptTokens: 250000, + PromptTokens: 280000, CompletionTokens: 30000, - TotalTokens: 280000, // Above 272k + TotalTokens: 310000, // input is above 272k PromptTokensDetails: &schemas.ChatPromptTokensDetails{ CachedReadTokens: 50000, }, @@ -575,11 +593,11 @@ func TestComputeTextCost_272kTierWithCacheRead(t *testing.T) { cost := computeTextCost(&p, usage, serviceTier{}) - // Non-cached input: (250000-50000) * 0.000009 = 200000 * 0.000009 = 1.80 + // Non-cached input: (280000-50000) * 0.000009 = 230000 * 0.000009 = 2.07 // Cached read: 50000 * 0.0000009 = 0.045 // Output: 30000 * 0.000045 = 1.35 - // Total: 1.80 + 0.045 + 1.35 = 3.195 - assert.InDelta(t, 3.195, cost, 1e-9) + // Total: 2.07 + 0.045 + 1.35 = 3.465 + assert.InDelta(t, 3.465, cost, 1e-9) } func TestComputeTextCost_SearchQueryCost(t *testing.T) { @@ -643,6 +661,21 @@ func TestComputeEmbeddingCost_Basic(t *testing.T) { assert.InDelta(t, 0.0005, cost, 1e-12) } +func TestComputeEmbeddingCost_TotalAbove200kButInputBelow200kUsesBaseRate(t *testing.T) { + p := configstoreTables.TableModelPricing{ + InputCostPerToken: bifrost.Ptr(0.000003), + InputCostPerTokenAbove200kTokens: bifrost.Ptr(0.000006), + } + usage := &schemas.BifrostLLMUsage{ + PromptTokens: 180000, + TotalTokens: 210000, + } + + cost := computeEmbeddingCost(&p, usage, serviceTier{}) + + assert.InDelta(t, 180000*0.000003, cost, 1e-9) +} + func TestComputeEmbeddingCost_NilUsage(t *testing.T) { p := configstoreTables.TableModelPricing{InputCostPerToken: new(0.0000001)} assert.Equal(t, 0.0, computeEmbeddingCost(&p, nil, serviceTier{})) @@ -667,6 +700,21 @@ func TestComputeRerankCost_Basic(t *testing.T) { assert.InDelta(t, 0.0022, cost, 1e-12) } +func TestComputeRerankCost_TotalAbove200kButInputBelow200kUsesBaseRate(t *testing.T) { + p := chatPricing(0.000003, 0.000015) + p.InputCostPerTokenAbove200kTokens = bifrost.Ptr(0.000006) + p.OutputCostPerTokenAbove200kTokens = bifrost.Ptr(0.00003) + usage := &schemas.BifrostLLMUsage{ + PromptTokens: 180000, + CompletionTokens: 30000, + TotalTokens: 210000, + } + + cost := computeRerankCost(&p, usage, serviceTier{}) + + assert.InDelta(t, 180000*0.000003+30000*0.000015, cost, 1e-9) +} + func TestComputeRerankCost_WithSearchCost(t *testing.T) { p := configstoreTables.TableModelPricing{ InputCostPerToken: bifrost.Ptr(0.0), @@ -760,6 +808,21 @@ func TestComputeSpeechCost_TokenFallback(t *testing.T) { assert.InDelta(t, 0.0125, cost, 1e-12) } +func TestComputeSpeechCost_TotalAbove200kButInputBelow200kUsesBaseRate(t *testing.T) { + p := chatPricing(0.000003, 0.000015) + p.InputCostPerTokenAbove200kTokens = bifrost.Ptr(0.000006) + p.OutputCostPerTokenAbove200kTokens = bifrost.Ptr(0.00003) + usage := &schemas.BifrostLLMUsage{ + PromptTokens: 180000, + CompletionTokens: 30000, + TotalTokens: 210000, + } + + cost := computeSpeechCost(&p, usage, nil, 0, serviceTier{}) + + assert.InDelta(t, 180000*0.000003+30000*0.000015, cost, 1e-9) +} + func TestComputeSpeechCost_NilUsageNilSeconds(t *testing.T) { p := chatPricing(0.000005, 0.000015) assert.Equal(t, 0.0, computeSpeechCost(&p, nil, nil, 0, serviceTier{})) @@ -817,6 +880,21 @@ func TestComputeTranscriptionCost_TokenFallback(t *testing.T) { assert.InDelta(t, 0.008, cost, 1e-12) } +func TestComputeTranscriptionCost_TotalAbove200kButInputBelow200kUsesBaseRate(t *testing.T) { + p := chatPricing(0.000003, 0.000015) + p.InputCostPerTokenAbove200kTokens = bifrost.Ptr(0.000006) + p.OutputCostPerTokenAbove200kTokens = bifrost.Ptr(0.00003) + usage := &schemas.BifrostLLMUsage{ + PromptTokens: 180000, + CompletionTokens: 30000, + TotalTokens: 210000, + } + + cost := computeTranscriptionCost(&p, usage, nil, nil, serviceTier{}) + + assert.InDelta(t, 180000*0.000003+30000*0.000015, cost, 1e-9) +} + func TestComputeTranscriptionCost_TokenDetailsPreferredOverDuration(t *testing.T) { // STT: audio token details present → uses tokens, not per-second p := configstoreTables.TableModelPricing{ @@ -903,6 +981,47 @@ func TestComputeImageCost_TokenBased(t *testing.T) { assert.InDelta(t, 0.0125, cost, 1e-12) } +func TestComputeImageCost_TotalAbove200kButInputBelow200kUsesBaseRate(t *testing.T) { + p := chatPricing(0.000003, 0.000015) + p.InputCostPerTokenAbove200kTokens = bifrost.Ptr(0.000006) + p.OutputCostPerTokenAbove200kTokens = bifrost.Ptr(0.00003) + usage := &schemas.ImageUsage{ + InputTokens: 180000, + OutputTokens: 30000, + TotalTokens: 210000, + } + + cost := computeImageCost(&p, usage, "", "", serviceTier{}) + + assert.InDelta(t, 180000*0.000003+30000*0.000015, cost, 1e-9) +} + +func TestComputeImageCost_DerivesTierTokensFromTotalMinusOutputWhenInputMissing(t *testing.T) { + p := chatPricing(0.000003, 0.000015) + p.OutputCostPerTokenAbove200kTokens = bifrost.Ptr(0.00003) + usage := &schemas.ImageUsage{ + OutputTokens: 30000, + TotalTokens: 240000, // derived input = 210000, so output uses long-context rate + } + + cost := computeImageCost(&p, usage, "", "", serviceTier{}) + + assert.InDelta(t, 30000*0.00003, cost, 1e-9) +} + +func TestComputeImageCost_DoesNotUseBareTotalTokensAsInputTierTokens(t *testing.T) { + p := chatPricing(0.000003, 0.000015) + p.OutputCostPerImage = bifrost.Ptr(0.05) + p.OutputCostPerTokenAbove200kTokens = bifrost.Ptr(0.00003) + usage := &schemas.ImageUsage{ + TotalTokens: 210000, // no input/output split; total includes output, so do not use it as input + } + + cost := computeImageCost(&p, usage, "", "", serviceTier{}) + + assert.InDelta(t, 0.05, cost, 1e-9) +} + func TestComputeImageCost_TokenBasedWithDetails(t *testing.T) { p := configstoreTables.TableModelPricing{ InputCostPerToken: bifrost.Ptr(0.000005), @@ -1071,6 +1190,21 @@ func TestComputeVideoCost_DurationBased(t *testing.T) { assert.InDelta(t, 0.0305, cost, 1e-12) } +func TestComputeVideoCost_TotalAbove200kButInputBelow200kUsesBaseRate(t *testing.T) { + p := chatPricing(0.000003, 0.000015) + p.InputCostPerTokenAbove200kTokens = bifrost.Ptr(0.000006) + p.OutputCostPerTokenAbove200kTokens = bifrost.Ptr(0.00003) + usage := &schemas.BifrostLLMUsage{ + PromptTokens: 180000, + CompletionTokens: 30000, + TotalTokens: 210000, + } + + cost := computeVideoCost(&p, usage, nil, serviceTier{}) + + assert.InDelta(t, 180000*0.000003+30000*0.000015, cost, 1e-9) +} + func TestComputeVideoCost_OutputCostPerSecondFallback(t *testing.T) { p := configstoreTables.TableModelPricing{ InputCostPerToken: bifrost.Ptr(0.0), @@ -1587,6 +1721,80 @@ func TestGetPricing_BedrockAddsAnthropicPrefix(t *testing.T) { assert.Equal(t, 0.000003, derefF(p.InputCostPerToken)) } +func TestGetPricing_BedrockAddsOpenAIPrefix(t *testing.T) { + s := testStoreWithPricing(map[string]configstoreTables.TableModelPricing{ + makeKey("openai.gpt-oss-120b", "bedrock", "chat"): chatPricing(0.00000015, 0.0000006), + }) + p := s.resolvePricing(schemas.RoutingInfo{Provider: "bedrock", Model: "gpt-oss-120b"}, schemas.ChatCompletionRequest, LookupScopes{Provider: "bedrock"}) + require.NotNil(t, p) + assert.Equal(t, 0.00000015, derefF(p.InputCostPerToken)) +} + +func TestGetPricing_BedrockAddsGooglePrefix(t *testing.T) { + s := testStoreWithPricing(map[string]configstoreTables.TableModelPricing{ + makeKey("google.gemma-4-31b", "bedrock", "chat"): chatPricing(0.00000014, 0.0000004), + }) + p := s.resolvePricing(schemas.RoutingInfo{Provider: "bedrock", Model: "gemma-4-31b"}, schemas.ChatCompletionRequest, LookupScopes{Provider: "bedrock"}) + require.NotNil(t, p) + assert.Equal(t, 0.00000014, derefF(p.InputCostPerToken)) +} + +func TestGetPricing_BedrockAddsXAIPrefix(t *testing.T) { + s := testStoreWithPricing(map[string]configstoreTables.TableModelPricing{ + makeKey("xai.grok-4.3", "bedrock", "chat"): chatPricing(0.00000125, 0.0000025), + }) + p := s.resolvePricing(schemas.RoutingInfo{Provider: "bedrock", Model: "grok-4.3"}, schemas.ChatCompletionRequest, LookupScopes{Provider: "bedrock"}) + require.NotNil(t, p) + assert.Equal(t, 0.00000125, derefF(p.InputCostPerToken)) +} + +func TestGetPricing_BedrockMantleFallsBackToBedrock(t *testing.T) { + s := testStoreWithPricing(map[string]configstoreTables.TableModelPricing{ + makeKey("openai.gpt-oss-120b", "bedrock", "chat"): chatPricing(0.00000015, 0.0000006), + }) + p := s.resolvePricing(schemas.RoutingInfo{Provider: "bedrock_mantle", Model: "openai.gpt-oss-120b"}, schemas.ChatCompletionRequest, LookupScopes{Provider: "bedrock_mantle"}) + require.NotNil(t, p) + assert.Equal(t, 0.00000015, derefF(p.InputCostPerToken)) +} + +func TestGetPricing_BedrockMantleResponsesFallsBackToBedrockChat(t *testing.T) { + s := testStoreWithPricing(map[string]configstoreTables.TableModelPricing{ + makeKey("openai.gpt-oss-120b", "bedrock", "chat"): chatPricing(0.00000015, 0.0000006), + }) + // bedrock_mantle provider + responses request → try bedrock + responses → try bedrock + chat + p := s.resolvePricing(schemas.RoutingInfo{Provider: "bedrock_mantle", Model: "openai.gpt-oss-120b"}, schemas.ResponsesRequest, LookupScopes{Provider: "bedrock_mantle"}) + require.NotNil(t, p) + assert.Equal(t, 0.00000015, derefF(p.InputCostPerToken)) +} + +func TestGetPricing_BedrockMantleAddsAnthropicPrefix(t *testing.T) { + s := testStoreWithPricing(map[string]configstoreTables.TableModelPricing{ + makeKey("anthropic.claude-3-5-sonnet-20241022-v2:0", "bedrock", "chat"): chatPricing(0.000003, 0.000015), + }) + p := s.resolvePricing(schemas.RoutingInfo{Provider: "bedrock_mantle", Model: "claude-3-5-sonnet-20241022-v2:0"}, schemas.ChatCompletionRequest, LookupScopes{Provider: "bedrock_mantle"}) + require.NotNil(t, p) + assert.Equal(t, 0.000003, derefF(p.InputCostPerToken)) +} + +func TestGetPricing_BedrockMantleAddsOpenAIPrefix(t *testing.T) { + s := testStoreWithPricing(map[string]configstoreTables.TableModelPricing{ + makeKey("openai.gpt-oss-120b", "bedrock", "chat"): chatPricing(0.00000015, 0.0000006), + }) + // bedrock_mantle folds onto bedrock, then the openai. prefix retry fires + p := s.resolvePricing(schemas.RoutingInfo{Provider: "bedrock_mantle", Model: "gpt-oss-120b"}, schemas.ChatCompletionRequest, LookupScopes{Provider: "bedrock_mantle"}) + require.NotNil(t, p) + assert.Equal(t, 0.00000015, derefF(p.InputCostPerToken)) +} + +func TestGetPricing_BedrockMantleResponsesAddsOpenAIPrefix(t *testing.T) { + s := testStoreWithPricing(map[string]configstoreTables.TableModelPricing{ + makeKey("openai.gpt-5.5", "bedrock", "responses"): chatPricing(0.0000055, 0.000033), + }) + p := s.resolvePricing(schemas.RoutingInfo{Provider: "bedrock_mantle", Model: "gpt-5.5"}, schemas.ResponsesRequest, LookupScopes{Provider: "bedrock_mantle"}) + require.NotNil(t, p) + assert.Equal(t, 0.0000055, derefF(p.InputCostPerToken)) +} + func TestGetPricing_ResponsesFallsBackToChat(t *testing.T) { s := testStoreWithPricing(map[string]configstoreTables.TableModelPricing{ makeKey("gpt-4o", "openai", "chat"): chatPricing(0.000005, 0.000015), @@ -1758,15 +1966,15 @@ func TestCalculateCost_200kTier_EndToEnd(t *testing.T) { }) resp := makeChatResponse(schemas.Bedrock, "anthropic.claude-3-5-sonnet-20240620-v1:0", &schemas.BifrostLLMUsage{ - PromptTokens: 190000, + PromptTokens: 210000, CompletionTokens: 20000, - TotalTokens: 210000, // Above 200k + TotalTokens: 230000, // input is above 200k }) cost := s.CalculateCost(resp, nil) // Tiered rate: input=0.000006, output=0.00003 - // 190000*0.000006 + 20000*0.00003 = 1.14 + 0.6 = 1.74 - assert.InDelta(t, 1.74, cost, 1e-9) + // 210000*0.000006 + 20000*0.00003 = 1.26 + 0.6 = 1.86 + assert.InDelta(t, 1.86, cost, 1e-9) } func TestCalculateCost_272kTier_EndToEnd(t *testing.T) { @@ -1788,15 +1996,15 @@ func TestCalculateCost_272kTier_EndToEnd(t *testing.T) { }) resp := makeChatResponse(schemas.Anthropic, "claude-3-7-sonnet", &schemas.BifrostLLMUsage{ - PromptTokens: 250000, + PromptTokens: 280000, CompletionTokens: 30000, - TotalTokens: 280000, // Above 272k + TotalTokens: 310000, // input is above 272k }) cost := s.CalculateCost(resp, nil) // Tiered rate: input=0.000009, output=0.000045 - // 250000*0.000009 + 30000*0.000045 = 2.25 + 1.35 = 3.60 - assert.InDelta(t, 3.60, cost, 1e-9) + // 280000*0.000009 + 30000*0.000045 = 2.52 + 1.35 = 3.87 + assert.InDelta(t, 3.87, cost, 1e-9) } func TestCalculateCost_272kTier_CacheReadFallbackChain(t *testing.T) { @@ -1817,20 +2025,20 @@ func TestCalculateCost_272kTier_CacheReadFallbackChain(t *testing.T) { }) resp := makeChatResponse(schemas.Anthropic, "claude-3-7-sonnet", &schemas.BifrostLLMUsage{ - PromptTokens: 250000, + PromptTokens: 280000, CompletionTokens: 30000, - TotalTokens: 280000, + TotalTokens: 310000, PromptTokensDetails: &schemas.ChatPromptTokensDetails{ CachedReadTokens: 50000, }, }) cost := s.CalculateCost(resp, nil) - // Non-cached input: (250000-50000) * 0.000009 = 200000 * 0.000009 = 1.80 + // Non-cached input: (280000-50000) * 0.000009 = 230000 * 0.000009 = 2.07 // Cached read (272k rate): 50000 * 0.0000009 = 0.045 // Output: 30000 * 0.000045 = 1.35 - // Total: 1.80 + 0.045 + 1.35 = 3.195 - assert.InDelta(t, 3.195, cost, 1e-9) + // Total: 2.07 + 0.045 + 1.35 = 3.465 + assert.InDelta(t, 3.465, cost, 1e-9) } // ========================================================================= @@ -1881,15 +2089,15 @@ func TestComputeTextCost_Priority272kTier(t *testing.T) { p.OutputCostPerTokenAbove272kTokensPriority = new(0.00006) usage := &schemas.BifrostLLMUsage{ - PromptTokens: 250000, + PromptTokens: 280000, CompletionTokens: 30000, - TotalTokens: 280000, + TotalTokens: 310000, } cost := computeTextCost(&p, usage, serviceTier{isPriority: true}) - // Uses 272k priority rates: 250000*0.000012 + 30000*0.00006 = 3.00 + 1.80 = 4.80 - assert.InDelta(t, 4.80, cost, 1e-9) + // Uses 272k priority rates: 280000*0.000012 + 30000*0.00006 = 3.36 + 1.80 = 5.16 + assert.InDelta(t, 5.16, cost, 1e-9) } func TestComputeTextCost_Priority272kTierFallsBackToNonPriority272k(t *testing.T) { @@ -1899,15 +2107,15 @@ func TestComputeTextCost_Priority272kTierFallsBackToNonPriority272k(t *testing.T p.OutputCostPerTokenAbove272kTokens = new(0.000045) usage := &schemas.BifrostLLMUsage{ - PromptTokens: 250000, + PromptTokens: 280000, CompletionTokens: 30000, - TotalTokens: 280000, + TotalTokens: 310000, } cost := computeTextCost(&p, usage, serviceTier{isPriority: true}) - // Falls back to non-priority 272k rate: 250000*0.000009 + 30000*0.000045 = 2.25 + 1.35 = 3.60 - assert.InDelta(t, 3.60, cost, 1e-9) + // Falls back to non-priority 272k rate: 280000*0.000009 + 30000*0.000045 = 2.52 + 1.35 = 3.87 + assert.InDelta(t, 3.87, cost, 1e-9) } func TestComputeTextCost_PriorityCacheReadRate(t *testing.T) { diff --git a/framework/modelcatalog/datasheet/overrides_test.go b/framework/modelcatalog/datasheet/overrides_test.go index 79a1c8d819e..7de05d2703e 100644 --- a/framework/modelcatalog/datasheet/overrides_test.go +++ b/framework/modelcatalog/datasheet/overrides_test.go @@ -33,6 +33,7 @@ func newTestStore() *Store { supportedResponseTypes: map[string][]string{}, supportedParams: map[string][]string{}, datasheetByProvider: map[schemas.ModelProvider][]string{}, + deprecatedByProvider: map[schemas.ModelProvider][]string{}, } } diff --git a/framework/modelcatalog/datasheet/params.go b/framework/modelcatalog/datasheet/params.go index 33c8cd7aa07..905da526edc 100644 --- a/framework/modelcatalog/datasheet/params.go +++ b/framework/modelcatalog/datasheet/params.go @@ -15,7 +15,6 @@ import ( "github.com/maximhq/bifrost/core/schemas" configstoreTables "github.com/maximhq/bifrost/framework/configstore/tables" "github.com/tidwall/gjson" - "gorm.io/gorm" ) // LoadModelParamsFromDB bulk-loads model parameters from the DB into the @@ -76,19 +75,14 @@ func (s *Store) SyncModelParamsFromURL(ctx context.Context) error { } if s.configStore != nil { - err = s.configStore.ExecuteTransaction(ctx, func(tx *gorm.DB) error { - for model, data := range paramsData { - params := &configstoreTables.TableModelParameters{ - Model: model, - Data: string(data), - } - if err := s.configStore.UpsertModelParameters(ctx, params, tx); err != nil { - return fmt.Errorf("failed to upsert model parameters for model %s: %w", model, err) - } - } - return nil - }) - if err != nil { + records := make([]configstoreTables.TableModelParameters, 0, len(paramsData)) + for model, data := range paramsData { + records = append(records, configstoreTables.TableModelParameters{ + Model: model, + Data: string(data), + }) + } + if err := s.configStore.UpsertModelParametersBatch(ctx, records); err != nil { return fmt.Errorf("failed to sync model parameters to database: %w", err) } } diff --git a/framework/modelcatalog/datasheet/store.go b/framework/modelcatalog/datasheet/store.go index 1ce5f01cf70..89e46171b11 100644 --- a/framework/modelcatalog/datasheet/store.go +++ b/framework/modelcatalog/datasheet/store.go @@ -61,6 +61,7 @@ type Store struct { supportedResponseTypes map[string][]string // model → [chat_completion, responses, …] supportedParams map[string][]string // model → [temperature, top_p, …] datasheetByProvider map[schemas.ModelProvider][]string // rebuilt every reload + deprecatedByProvider map[schemas.ModelProvider][]string // rebuilt every reload // Overrides under their own mutex: writes here don't block pricing reads // (the hot CalculateCost path takes mu.RLock and overridesMu.RLock @@ -91,6 +92,7 @@ func New(configStore configstore.ConfigStore, logger schemas.Logger, cfg Config) supportedResponseTypes: make(map[string][]string), supportedParams: make(map[string][]string), datasheetByProvider: make(map[schemas.ModelProvider][]string), + deprecatedByProvider: make(map[schemas.ModelProvider][]string), url: cfg.URL, modelParametersURL: cfg.ModelParametersURL, syncInterval: cfg.SyncInterval, @@ -282,6 +284,21 @@ func (s *Store) DatasheetModelsForProvider(provider schemas.ModelProvider) []str return out } +// DeprecatedDatasheetModelsForProvider returns deprecated models from the +// datasheet view for provider. Deprecated models may disappear from provider +// list-models APIs but must remain visible in Bifrost catalog listings. +func (s *Store) DeprecatedDatasheetModelsForProvider(provider schemas.ModelProvider) []string { + s.mu.RLock() + defer s.mu.RUnlock() + models, ok := s.deprecatedByProvider[provider] + if !ok { + return nil + } + out := make([]string, len(models)) + copy(out, models) + return out +} + // DatasheetProviders returns every provider that has at least one pricing // row in the datasheet view. Composer unions this with live + keyconfig to // enumerate "all known providers" for GetProvidersForModel. @@ -416,17 +433,19 @@ func NewTestStore(baseModelIndex map[string]string) *Store { supportedResponseTypes: make(map[string][]string), supportedParams: make(map[string][]string), datasheetByProvider: make(map[schemas.ModelProvider][]string), + deprecatedByProvider: make(map[schemas.ModelProvider][]string), } } // --- Internal: rebuild the datasheet view from current pricingData --- -// rebuildDatasheetViewUnsafe regenerates baseModelIndex and datasheetByProvider -// from pricingData. Caller MUST hold s.mu write-lock. Called after every -// pricingData mutation in sync.go / params.go. +// rebuildDatasheetViewUnsafe regenerates baseModelIndex, datasheetByProvider, +// and deprecatedByProvider from pricingData. Caller MUST hold s.mu write-lock. +// Called after every pricingData mutation in sync.go / params.go. func (s *Store) rebuildDatasheetViewUnsafe() { s.baseModelIndex = make(map[string]string) providerModels := make(map[schemas.ModelProvider]map[string]struct{}) + deprecatedModels := make(map[schemas.ModelProvider]map[string]struct{}) for _, pricing := range s.pricingData { normalized := schemas.ModelProvider(normalizeProvider(pricing.Provider)) @@ -434,6 +453,12 @@ func (s *Store) rebuildDatasheetViewUnsafe() { providerModels[normalized] = make(map[string]struct{}) } providerModels[normalized][pricing.Model] = struct{}{} + if pricing.IsDeprecated { + if deprecatedModels[normalized] == nil { + deprecatedModels[normalized] = make(map[string]struct{}) + } + deprecatedModels[normalized][pricing.Model] = struct{}{} + } if pricing.BaseModel != "" { s.baseModelIndex[pricing.Model] = pricing.BaseModel @@ -449,4 +474,14 @@ func (s *Store) rebuildDatasheetViewUnsafe() { slices.Sort(models) s.datasheetByProvider[provider] = models } + + s.deprecatedByProvider = make(map[schemas.ModelProvider][]string, len(deprecatedModels)) + for provider, modelSet := range deprecatedModels { + models := make([]string, 0, len(modelSet)) + for m := range modelSet { + models = append(models, m) + } + slices.Sort(models) + s.deprecatedByProvider[provider] = models + } } diff --git a/framework/modelcatalog/datasheet/store_test.go b/framework/modelcatalog/datasheet/store_test.go new file mode 100644 index 00000000000..5807d1d34a0 --- /dev/null +++ b/framework/modelcatalog/datasheet/store_test.go @@ -0,0 +1,63 @@ +package datasheet + +import ( + "slices" + "testing" + + "github.com/maximhq/bifrost/core/schemas" + configstoreTables "github.com/maximhq/bifrost/framework/configstore/tables" +) + +func TestDeprecatedDatasheetModelsForProviderUsesRebuiltIndex(t *testing.T) { + s := NewTestStore(nil) + s.mu.Lock() + s.pricingData[makeKey("deprecated-b", "openai", "chat")] = configstoreTables.TableModelPricing{ + Model: "deprecated-b", + Provider: "openai", + Mode: "chat", + IsDeprecated: true, + } + s.pricingData[makeKey("deprecated-a", "openai", "chat")] = configstoreTables.TableModelPricing{ + Model: "deprecated-a", + Provider: "openai", + Mode: "chat", + IsDeprecated: true, + } + s.pricingData[makeKey("deprecated-a", "openai", "responses")] = configstoreTables.TableModelPricing{ + Model: "deprecated-a", + Provider: "openai", + Mode: "responses", + IsDeprecated: true, + } + s.pricingData[makeKey("active", "openai", "chat")] = configstoreTables.TableModelPricing{ + Model: "active", + Provider: "openai", + Mode: "chat", + } + s.pricingData[makeKey("deprecated-vertex", "vertex_ai", "chat")] = configstoreTables.TableModelPricing{ + Model: "deprecated-vertex", + Provider: "vertex_ai", + Mode: "chat", + IsDeprecated: true, + } + s.rebuildDatasheetViewUnsafe() + s.mu.Unlock() + + got := s.DeprecatedDatasheetModelsForProvider(schemas.OpenAI) + want := []string{"deprecated-a", "deprecated-b"} + if !slices.Equal(got, want) { + t.Fatalf("expected deprecated OpenAI models %v, got %v", want, got) + } + + got[0] = "mutated" + got = s.DeprecatedDatasheetModelsForProvider(schemas.OpenAI) + if !slices.Equal(got, want) { + t.Fatalf("expected defensive copy from index %v, got %v", want, got) + } + + got = s.DeprecatedDatasheetModelsForProvider(schemas.Vertex) + want = []string{"deprecated-vertex"} + if !slices.Equal(got, want) { + t.Fatalf("expected deprecated Vertex models %v, got %v", want, got) + } +} diff --git a/framework/modelcatalog/datasheet/types.go b/framework/modelcatalog/datasheet/types.go index 80aa5d71380..2616bf8694f 100644 --- a/framework/modelcatalog/datasheet/types.go +++ b/framework/modelcatalog/datasheet/types.go @@ -45,6 +45,7 @@ type Entry struct { MaxInputTokens *int `json:"max_input_tokens,omitempty"` MaxOutputTokens *int `json:"max_output_tokens,omitempty"` Architecture *schemas.Architecture `json:"architecture,omitempty"` + IsDeprecated bool `json:"is_deprecated,omitempty"` // AdditionalAttributes carries editorial metadata stored on the pricing // row (e.g. description). Populated from the DB read path only; the @@ -294,7 +295,7 @@ type customPricingData struct { } // modelParametersParseResult is the parsed result type used by -// buildSupportedOutputsIndex (consumed by params.go's applyModelParameters). +// extractSupportedParams (consumed by params.go's applyModelParameters). type modelParametersParseResult struct { Mode *string `json:"mode,omitempty"` SupportedEndpoints []string `json:"supported_endpoints,omitempty"` @@ -307,6 +308,7 @@ type modelParametersParseResult struct { SupportsToolChoice *bool `json:"supports_tool_choice,omitempty"` SupportsReasoning *bool `json:"supports_reasoning,omitempty"` SupportsResponseSchema *bool `json:"supports_response_schema,omitempty"` + SupportsReasoningWithToolCalls *bool `json:"supports_reasoning_with_tool_calls,omitempty"` SupportsServiceTier *bool `json:"supports_service_tier,omitempty"` SupportsPromptCaching *bool `json:"supports_prompt_caching,omitempty"` SupportsWebSearch *bool `json:"supports_web_search,omitempty"` @@ -476,6 +478,9 @@ func extractSupportedParams(parsed *modelParametersParseResult) []string { if parsed.SupportsReasoning != nil && *parsed.SupportsReasoning { addParam("reasoning") } + if parsed.SupportsReasoningWithToolCalls == nil || *parsed.SupportsReasoningWithToolCalls { + addParam("reasoning_with_tool_calls") + } if parsed.SupportsResponseSchema != nil && *parsed.SupportsResponseSchema { addParam("response_format") addParam("text") @@ -548,6 +553,7 @@ func convertEntryToTablePricing(modelKey string, entry Entry) configstoreTables. MaxInputTokens: entry.MaxInputTokens, MaxOutputTokens: entry.MaxOutputTokens, Architecture: entry.Architecture, + IsDeprecated: entry.IsDeprecated, InputCostPerToken: entry.InputCostPerToken, OutputCostPerToken: entry.OutputCostPerToken, @@ -705,6 +711,7 @@ func convertTablePricingToEntry(pricing *configstoreTables.TableModelPricing) *E MaxInputTokens: pricing.MaxInputTokens, MaxOutputTokens: pricing.MaxOutputTokens, Architecture: pricing.Architecture, + IsDeprecated: pricing.IsDeprecated, AdditionalAttributes: pricing.AdditionalAttributes, Options: options, } diff --git a/framework/modelcatalog/main.go b/framework/modelcatalog/main.go index 7d4f41aa4ac..bdfe1913a17 100644 --- a/framework/modelcatalog/main.go +++ b/framework/modelcatalog/main.go @@ -367,21 +367,6 @@ func (mc *ModelCatalog) ForceReloadPricing(ctx context.Context) error { } }() - // MCP library sync runs alongside but is non-fatal: a failure here must not - // block a pricing/params force-reload. It is logged and the last-sync - // timestamp is only advanced on success. - wg.Add(1) - go func() { - defer wg.Done() - if err := mc.syncMCPLibrary(ctx); err != nil { - mc.logger.Warn("MCP library sync during force-reload failed: %v", err) - return - } - mc.syncMu.Lock() - mc.lastMCPLibrarySyncedAt = time.Now() - mc.syncMu.Unlock() - }() - wg.Wait() if pricingErr != nil { return pricingErr diff --git a/framework/modelcatalog/models.go b/framework/modelcatalog/models.go index b28454669c6..f36c31aa510 100644 --- a/framework/modelcatalog/models.go +++ b/framework/modelcatalog/models.go @@ -9,6 +9,17 @@ import ( "github.com/maximhq/bifrost/framework/configstore" ) +// providersWithPartialListModels enumerates providers whose /v1/models response +// is a strict subset of their callable catalog — Perplexity lists only +// responses-API models and omits the Sonar chat family, which is still callable +// via /chat/completions. For these, the datasheet (not the live list) is the +// authoritative superset, so live must be unioned with the full allowed +// datasheet set rather than only the deprecated backfill. +var providersWithPartialListModels = map[schemas.ModelProvider]bool{ + schemas.Perplexity: true, + schemas.Vertex: true, +} + // GetModelsForProvider returns the effective allowed model set for the // provider. Filtered live entries are authoritative when present (they were // pre-gated by ListModelsPipeline against the key's allow/block/aliases); @@ -20,6 +31,16 @@ func (mc *ModelCatalog) GetModelsForProvider(provider schemas.ModelProvider) []s var out []string if liveModels := mc.live.ModelsForProvider(provider); len(liveModels) > 0 { out = liveModels + // Datasheet models to reconcile on top of the live list: normally just + // deprecated ones (dropped from list-models but still callable). For + // providers whose list-models is a partial subset, use the full + // datasheet so callable-but-unlisted models (e.g. Perplexity's Sonar + // chat family) aren't shadowed by the incomplete live list. + datasheetModelsToAppend := mc.datasheet.DeprecatedDatasheetModelsForProvider(provider) + if providersWithPartialListModels[provider] { + datasheetModelsToAppend = mc.datasheet.DatasheetModelsForProvider(provider) + } + out = mc.appendAllowedDatasheetModels(out, datasheetModelsToAppend, allowed, blacklisted) } else if datasheetModels := mc.datasheet.DatasheetModelsForProvider(provider); len(datasheetModels) > 0 && allowed != nil { out = make([]string, 0, len(datasheetModels)) for _, m := range datasheetModels { @@ -69,6 +90,31 @@ func (mc *ModelCatalog) GetModelsForProvider(provider schemas.ModelProvider) []s return out } +func (mc *ModelCatalog) appendAllowedDatasheetModels(out []string, models []string, allowed schemas.WhiteList, blacklisted schemas.BlackList) []string { + if len(models) == 0 { + return out + } + seen := make(map[string]struct{}, len(out)) + for _, m := range out { + seen[m] = struct{}{} + } + for _, m := range models { + if _, ok := seen[m]; ok { + continue + } + if blacklisted.IsBlocked(m) { + continue + } + if allowed != nil && !allowed.IsAllowed(m) { + continue + } + seen[m] = struct{}{} + out = append(out, m) + } + slices.Sort(out) + return out +} + // GetUnfilteredModelsForProvider returns the raw catalog view (no gate // applied): union of live unfiltered entries and the datasheet view. func (mc *ModelCatalog) GetUnfilteredModelsForProvider(provider schemas.ModelProvider) []string { @@ -257,6 +303,8 @@ func (mc *ModelCatalog) RefineModelForProvider(provider schemas.ModelProvider, m return mc.refineNestedProviderModel(provider, model) case schemas.Replicate: return mc.refineNestedProviderModel(provider, model) + case schemas.Perplexity: + return mc.refineNestedProviderModel(provider, model) } return model, nil } diff --git a/framework/modelcatalog/pool_test.go b/framework/modelcatalog/pool_test.go index b9dc4c5f0ff..d3216018e6e 100644 --- a/framework/modelcatalog/pool_test.go +++ b/framework/modelcatalog/pool_test.go @@ -1,10 +1,15 @@ package modelcatalog import ( + "os" + "path/filepath" "slices" "testing" "github.com/maximhq/bifrost/core/schemas" + "github.com/maximhq/bifrost/framework/modelcatalog/datasheet" + "github.com/maximhq/bifrost/framework/modelcatalog/keyconfig" + "github.com/maximhq/bifrost/framework/modelcatalog/live" ) // TestUpsertLiveFromResponse_NilRespIsNoop guards the API surface: handing a @@ -47,6 +52,154 @@ func TestUpsertLiveFromResponse_PopulatesFromResponse(t *testing.T) { } } +func TestGetModelsForProvider_IncludesDeprecatedDatasheetModelsWhenLiveExists(t *testing.T) { + pricingPath := filepath.Join(t.TempDir(), "pricing.json") + pricingJSON := []byte(`{ + "deprecated-model": {"provider":"openai","mode":"chat","base_model":"deprecated-model","is_deprecated":true}, + "current-model": {"provider":"openai","mode":"chat","base_model":"current-model"} + }`) + if err := os.WriteFile(pricingPath, pricingJSON, 0o600); err != nil { + t.Fatalf("write pricing testdata: %v", err) + } + + ds := datasheet.New(nil, nil, datasheet.Config{URL: "file://" + pricingPath}) + if err := ds.LoadFromURLIntoMemory(t.Context()); err != nil { + t.Fatalf("load pricing testdata: %v", err) + } + mc := NewTestCatalogWithDatasheet(ds) + mc.UpsertLive(schemas.OpenAI, "k1", false, []string{"live-model"}) + + got := mc.GetModelsForProvider(schemas.OpenAI) + slices.Sort(got) + want := []string{"deprecated-model", "live-model"} + if !slices.Equal(got, want) { + t.Errorf("GetModelsForProvider = %v, want %v", got, want) + } +} + +// deprecatedDatasheet returns a datasheet.Store loaded from a temp pricing +// file containing one deprecated model and one current model, both openai. +func deprecatedDatasheet(t *testing.T) *datasheet.Store { + t.Helper() + pricingPath := filepath.Join(t.TempDir(), "pricing.json") + pricingJSON := []byte(`{ + "deprecated-model": {"provider":"openai","mode":"chat","base_model":"deprecated-model","is_deprecated":true}, + "current-model": {"provider":"openai","mode":"chat","base_model":"current-model"} + }`) + if err := os.WriteFile(pricingPath, pricingJSON, 0o600); err != nil { + t.Fatalf("write pricing testdata: %v", err) + } + ds := datasheet.New(nil, nil, datasheet.Config{URL: "file://" + pricingPath}) + if err := ds.LoadFromURLIntoMemory(t.Context()); err != nil { + t.Fatalf("load pricing testdata: %v", err) + } + return ds +} + +// TestGetModelsForProvider_DeprecatedDatasheetModelsRespectAllowBlock pins the +// allow/block gating that appendAllowedDatasheetModels applies to deprecated +// datasheet models once a live entry exists. The unrestricted path is covered +// by TestGetModelsForProvider_IncludesDeprecatedDatasheetModelsWhenLiveExists; +// these sub-tests cover the keyconfig-restricted branches. +func TestGetModelsForProvider_DeprecatedDatasheetModelsRespectAllowBlock(t *testing.T) { + tests := []struct { + name string + keys []schemas.Key + want []string + }{ + { + // deprecated-model sits in the keyconfig blocklist → excluded even + // though the key otherwise allows everything (wildcard). + name: "blocklisted deprecated model is excluded", + keys: []schemas.Key{ + {ID: "k1", Enabled: ptrBool(true), Models: schemas.WhiteList{"*"}, BlacklistedModels: schemas.BlackList{"deprecated-model"}}, + }, + want: []string{"live-model"}, + }, + { + // explicit allowlist omits deprecated-model → excluded. + name: "deprecated model absent from allowlist is excluded", + keys: []schemas.Key{ + {ID: "k1", Enabled: ptrBool(true), Models: schemas.WhiteList{"live-model"}}, + }, + want: []string{"live-model"}, + }, + { + // explicit allowlist names deprecated-model → included. + name: "deprecated model present in allowlist is included", + keys: []schemas.Key{ + {ID: "k1", Enabled: ptrBool(true), Models: schemas.WhiteList{"live-model", "deprecated-model"}}, + }, + want: []string{"deprecated-model", "live-model"}, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + kc := keyconfig.New(nil) + kc.Replace(map[schemas.ModelProvider][]schemas.Key{schemas.OpenAI: tt.keys}) + mc := &ModelCatalog{ + datasheet: deprecatedDatasheet(t), + live: live.New(nil), + keyconf: kc, + done: make(chan struct{}), + } + mc.UpsertLive(schemas.OpenAI, "k1", false, []string{"live-model"}) + + got := mc.GetModelsForProvider(schemas.OpenAI) + slices.Sort(got) + if !slices.Equal(got, tt.want) { + t.Errorf("GetModelsForProvider = %v, want %v", got, tt.want) + } + }) + } +} + +// TestGetModelsForProvider_PartialListModelsProviderUnionsDatasheet pins the +// fix for Perplexity's partial /v1/models response: a callable-but-unlisted +// datasheet model (the Sonar chat family) must survive once a live entry +// exists, instead of being shadowed by the incomplete live list. The same +// non-deprecated datasheet model stays excluded for a normal provider, whose +// live list is treated as authoritative. +func TestGetModelsForProvider_PartialListModelsProviderUnionsDatasheet(t *testing.T) { + pricingPath := filepath.Join(t.TempDir(), "pricing.json") + pricingJSON := []byte(`{ + "sonar": {"provider":"perplexity","mode":"chat","base_model":"sonar"}, + "openai-unlisted": {"provider":"openai","mode":"chat","base_model":"openai-unlisted"} + }`) + if err := os.WriteFile(pricingPath, pricingJSON, 0o600); err != nil { + t.Fatalf("write pricing testdata: %v", err) + } + ds := datasheet.New(nil, nil, datasheet.Config{URL: "file://" + pricingPath}) + if err := ds.LoadFromURLIntoMemory(t.Context()); err != nil { + t.Fatalf("load pricing testdata: %v", err) + } + + // Perplexity: live lists only a responses-API model; the unlisted "sonar" + // chat model must still be unioned in from the datasheet. + mcPerplexity := NewTestCatalogWithDatasheet(ds) + mcPerplexity.UpsertLive(schemas.Perplexity, "k1", false, []string{"listed-responses-model"}) + gotPerplexity := mcPerplexity.GetModelsForProvider(schemas.Perplexity) + slices.Sort(gotPerplexity) + wantPerplexity := []string{"listed-responses-model", "sonar"} + if !slices.Equal(gotPerplexity, wantPerplexity) { + t.Errorf("GetModelsForProvider(Perplexity) = %v, want %v (unlisted datasheet model must be unioned)", gotPerplexity, wantPerplexity) + } + + // OpenAI is not a partial-list-models provider: its non-deprecated unlisted + // datasheet model stays shadowed by the authoritative live list. + mcOpenAI := NewTestCatalogWithDatasheet(ds) + mcOpenAI.UpsertLive(schemas.OpenAI, "k1", false, []string{"live-model"}) + gotOpenAI := mcOpenAI.GetModelsForProvider(schemas.OpenAI) + slices.Sort(gotOpenAI) + wantOpenAI := []string{"live-model"} + if !slices.Equal(gotOpenAI, wantOpenAI) { + t.Errorf("GetModelsForProvider(OpenAI) = %v, want %v (non-deprecated unlisted datasheet model must stay shadowed)", gotOpenAI, wantOpenAI) + } +} + +// ptrBool returns a pointer to b, for building schemas.Key fixtures. +func ptrBool(b bool) *bool { return &b } + // TestExtractModelIDs_StripsOwningProviderPrefix verifies the canonical // shape returned by every provider's ListModels — an ID prefixed with its // own provider key — gets reduced to a bare model name. diff --git a/framework/oauth2/sync.go b/framework/oauth2/sync.go index a46ae6e4f86..2bfefe8cd11 100644 --- a/framework/oauth2/sync.go +++ b/framework/oauth2/sync.go @@ -5,6 +5,7 @@ import ( "sync" "time" + bifrost "github.com/maximhq/bifrost/core" "github.com/maximhq/bifrost/core/schemas" ) @@ -15,11 +16,15 @@ type TokenRefreshWorker struct { lookAheadWindow time.Duration // How far ahead to look for expiring tokens stopCh chan struct{} stopOnce sync.Once + cancel context.CancelFunc logger schemas.Logger } // NewTokenRefreshWorker creates a new token refresh worker func NewTokenRefreshWorker(provider *OAuth2Provider, logger schemas.Logger) *TokenRefreshWorker { + if logger == nil { + logger = bifrost.NewNoOpLogger() + } if provider.configStore == nil { logger.Warn("config store is nil, skipping token refresh worker") return nil @@ -35,10 +40,10 @@ func NewTokenRefreshWorker(provider *OAuth2Provider, logger schemas.Logger) *Tok // Start begins the token refresh worker in a background goroutine func (w *TokenRefreshWorker) Start(ctx context.Context) { - go w.run(ctx) - if w.logger != nil { - w.logger.Info("Token refresh worker started") - } + runCtx, cancel := context.WithCancel(ctx) + w.cancel = cancel + go w.run(runCtx) + w.logger.Info("Token refresh worker started") } // Stop gracefully stops the token refresh worker. Safe to call multiple times @@ -46,10 +51,13 @@ func (w *TokenRefreshWorker) Start(ctx context.Context) { // can't panic by re-closing the channel. func (w *TokenRefreshWorker) Stop() { w.stopOnce.Do(func() { - close(w.stopCh) - if w.logger != nil { - w.logger.Info("Token refresh worker stopped") + // Cancel any in-flight refresh so a blocked DB call unwinds promptly, + // then signal run() to exit its ticker loop. + if w.cancel != nil { + w.cancel() } + close(w.stopCh) + w.logger.Info("Token refresh worker stopped") }) } @@ -80,52 +88,42 @@ func (w *TokenRefreshWorker) refreshExpiredTokens(ctx context.Context) { // Get tokens expiring before the threshold tokens, err := w.provider.configStore.GetExpiringOauthTokens(ctx, expiryThreshold) if err != nil { - if w.logger != nil { - w.logger.Error("Failed to get expiring tokens", "error", err) - } + w.logger.Error("Failed to get expiring tokens: %v", err) return } if len(tokens) == 0 { return } - - if w.logger != nil { - w.logger.Debug("Found expiring tokens to refresh: %d", len(tokens)) - } + w.logger.Debug("Found expiring tokens to refresh: %d", len(tokens)) // Refresh each expiring token for _, token := range tokens { // Find the oauth_config that references this token oauthConfig, err := w.provider.configStore.GetOauthConfigByTokenID(ctx, token.ID) if err != nil { - if w.logger != nil { - w.logger.Error("Failed to find oauth config for token: %s, error: %s", token.ID, err.Error()) - } + w.logger.Error("Failed to find oauth config for token: %s, error: %s", token.ID, err.Error()) continue } if oauthConfig == nil { - if w.logger != nil { - w.logger.Warn("No oauth config found for token: %s", token.ID) - } + w.logger.Warn("No oauth config found for token: %s", token.ID) continue } - // Attempt to refresh the token + // Attempt to refresh the token. Logged at Debug: transient failures + // (DNS, timeout, offline) recur on every tick and would spam the + // error log, while permanent rejections are already surfaced by the + // oauth_config status flipping to "expired" below. if err := w.provider.RefreshAccessToken(ctx, oauthConfig.ID); err != nil { - if w.logger != nil { - w.logger.Error("Failed to refresh token", "oauth_config_id", oauthConfig.ID, "error", err) - } + w.logger.Debug("Failed to refresh token: oauth_config_id: %s, error: %s", oauthConfig.ID, err.Error()) // Only mark as expired for permanent auth rejections (e.g. invalid_grant, 401). // Transient failures (DNS, timeout, offline) are skipped — the worker will // retry on the next tick and the connection heals automatically when online. w.provider.markExpiredIfPermanent(ctx, oauthConfig, err) } else { - if w.logger != nil { - w.logger.Debug("Successfully refreshed token: %s", oauthConfig.ID) - } + w.logger.Debug("Successfully refreshed token: %s", oauthConfig.ID) } } } @@ -154,16 +152,18 @@ type PerUserOAuthSweepWorker struct { orphanRetention time.Duration stopCh chan struct{} stopOnce sync.Once + cancel context.CancelFunc logger schemas.Logger } // NewPerUserOAuthSweepWorker creates a sweep worker with sensible defaults. // orphanRetention <= 0 disables the orphan-token sweep. func NewPerUserOAuthSweepWorker(provider *OAuth2Provider, orphanRetention time.Duration, logger schemas.Logger) *PerUserOAuthSweepWorker { + if logger == nil { + logger = bifrost.NewNoOpLogger() + } if provider == nil || provider.configStore == nil { - if logger != nil { - logger.Warn("per-user OAuth sweep worker not started: provider or config store is nil") - } + logger.Warn("per-user OAuth sweep worker not started: provider or config store is nil") return nil } return &PerUserOAuthSweepWorker{ @@ -178,21 +178,24 @@ func NewPerUserOAuthSweepWorker(provider *OAuth2Provider, orphanRetention time.D // Start begins the sweep worker in a background goroutine. func (w *PerUserOAuthSweepWorker) Start(ctx context.Context) { - go w.run(ctx) - if w.logger != nil { - w.logger.Info("Per-user OAuth sweep worker started (flow=%s, orphan=%s, retention=%s)", - w.flowSweepEvery, w.orphanSweepEvery, w.orphanRetention) - } + runCtx, cancel := context.WithCancel(ctx) + w.cancel = cancel + go w.run(runCtx) + w.logger.Info("Per-user OAuth sweep worker started (flow=%s, orphan=%s, retention=%s)", + w.flowSweepEvery, w.orphanSweepEvery, w.orphanRetention) } // Stop gracefully stops the sweep worker. sync.Once guards against double-close // panics when called from multiple shutdown paths. func (w *PerUserOAuthSweepWorker) Stop() { w.stopOnce.Do(func() { - close(w.stopCh) - if w.logger != nil { - w.logger.Info("Per-user OAuth sweep worker stopped") + // Cancel any in-flight sweep so a blocked DB call unwinds promptly, + // then signal run() to exit its ticker loop. + if w.cancel != nil { + w.cancel() } + close(w.stopCh) + w.logger.Info("Per-user OAuth sweep worker stopped") }) } @@ -223,12 +226,10 @@ func (w *PerUserOAuthSweepWorker) run(ctx context.Context) { func (w *PerUserOAuthSweepWorker) sweepExpiredFlows(ctx context.Context) { n, err := w.provider.configStore.DeleteExpiredOauthUserSessions(ctx) if err != nil { - if w.logger != nil { - w.logger.Error("per-user OAuth flow sweep failed: %v", err) - } + w.logger.Error("per-user OAuth flow sweep failed: %v", err) return } - if n > 0 && w.logger != nil { + if n > 0 { w.logger.Debug("per-user OAuth flow sweep removed %d expired pending flows", n) } } @@ -239,12 +240,10 @@ func (w *PerUserOAuthSweepWorker) sweepOrphanedTokens(ctx context.Context) { } n, err := w.provider.configStore.DeleteOrphanedOauthUserTokens(ctx, w.orphanRetention) if err != nil { - if w.logger != nil { - w.logger.Error("per-user OAuth orphan-token sweep failed: %v", err) - } + w.logger.Error("per-user OAuth orphan-token sweep failed: %v", err) return } - if n > 0 && w.logger != nil { + if n > 0 { w.logger.Info("per-user OAuth orphan-token sweep removed %d rows older than %s", n, w.orphanRetention) } } diff --git a/framework/streaming/responses.go b/framework/streaming/responses.go index c7bb8c7a6fa..e78ecebdfff 100644 --- a/framework/streaming/responses.go +++ b/framework/streaming/responses.go @@ -347,6 +347,15 @@ func deepCopyResponsesMessage(original schemas.ResponsesMessage) schemas.Respons } } + if original.ResponsesToolMessage.Caller != nil { + copyCaller := *original.ResponsesToolMessage.Caller + if original.ResponsesToolMessage.Caller.ToolID != nil { + copyToolID := *original.ResponsesToolMessage.Caller.ToolID + copyCaller.ToolID = ©ToolID + } + copy.ResponsesToolMessage.Caller = ©Caller + } + // Deep copy embedded tool call structs if original.ResponsesToolMessage.ResponsesFileSearchToolCall != nil { copyToolCall := *original.ResponsesToolMessage.ResponsesFileSearchToolCall @@ -400,6 +409,23 @@ func deepCopyResponsesMessage(original schemas.ResponsesMessage) schemas.Respons copy.ResponsesToolMessage.ResponsesComputerToolCallOutput = ©Output } + if original.ResponsesToolMessage.ResponsesWebFetchCall != nil { + copyCall := *original.ResponsesToolMessage.ResponsesWebFetchCall + if original.ResponsesToolMessage.ResponsesWebFetchCall.Document != nil { + docCopy := *original.ResponsesToolMessage.ResponsesWebFetchCall.Document + if original.ResponsesToolMessage.ResponsesWebFetchCall.Document.Source != nil { + srcCopy := *original.ResponsesToolMessage.ResponsesWebFetchCall.Document.Source + docCopy.Source = &srcCopy + } + if original.ResponsesToolMessage.ResponsesWebFetchCall.Document.Citations != nil { + citationsCopy := *original.ResponsesToolMessage.ResponsesWebFetchCall.Document.Citations + docCopy.Citations = &citationsCopy + } + copyCall.Document = &docCopy + } + copy.ResponsesToolMessage.ResponsesWebFetchCall = ©Call + } + if original.ResponsesToolMessage.ResponsesCodeInterpreterToolCall != nil { copyToolCall := *original.ResponsesToolMessage.ResponsesCodeInterpreterToolCall // Deep copy Outputs slice diff --git a/framework/streaming/responses_test.go b/framework/streaming/responses_test.go index 3685436d646..c48c88ecb8e 100644 --- a/framework/streaming/responses_test.go +++ b/framework/streaming/responses_test.go @@ -97,6 +97,40 @@ func TestBuildResponsesMessageRoutesParallelToolArgs(t *testing.T) { } } +func TestDeepCopyResponsesStreamResponseCopiesToolCaller(t *testing.T) { + original := &schemas.BifrostResponsesStreamResponse{ + Type: schemas.ResponsesStreamResponseTypeOutputItemDone, + Item: &schemas.ResponsesMessage{ + ID: schemas.Ptr("srvtoolu_fetch"), + Type: schemas.Ptr(schemas.ResponsesMessageTypeWebFetchCall), + ResponsesToolMessage: &schemas.ResponsesToolMessage{ + CallID: schemas.Ptr("srvtoolu_fetch"), + Caller: &schemas.ResponsesToolCaller{ + Type: "code_execution_20260120", + ToolID: schemas.Ptr("srvtoolu_code"), + }, + }, + }, + } + + copied := deepCopyResponsesStreamResponse(original) + if copied == nil || copied.Item == nil || copied.Item.ResponsesToolMessage == nil || copied.Item.ResponsesToolMessage.Caller == nil { + t.Fatalf("expected caller to be copied, got %#v", copied) + } + if copied.Item.ResponsesToolMessage.Caller == original.Item.ResponsesToolMessage.Caller { + t.Fatal("caller pointer was aliased") + } + if got := copied.Item.ResponsesToolMessage.Caller.Type; got != "code_execution_20260120" { + t.Fatalf("caller type = %q", got) + } + if copied.Item.ResponsesToolMessage.Caller.ToolID == nil || *copied.Item.ResponsesToolMessage.Caller.ToolID != "srvtoolu_code" { + t.Fatalf("caller tool id not preserved: %#v", copied.Item.ResponsesToolMessage.Caller) + } + if copied.Item.ResponsesToolMessage.Caller.ToolID == original.Item.ResponsesToolMessage.Caller.ToolID { + t.Fatal("caller tool id pointer was aliased") + } +} + // TestBuildResponsesMessageAccumulatesReasoningSummary verifies reasoning // summary deltas (no content index) concatenate into a single summary entry. func TestBuildResponsesMessageAccumulatesReasoningSummary(t *testing.T) { diff --git a/framework/temptoken/scope.go b/framework/temptoken/scope.go index e7cb789891a..bb9660f4d7b 100644 --- a/framework/temptoken/scope.go +++ b/framework/temptoken/scope.go @@ -23,6 +23,14 @@ const ( // flow endpoints. Bound resource_id is the headers flow ID. Parallel // of MCPAuthScopeName for the per-user-headers surface. MCPHeadersAuthScopeName = "mcp_headers_auth" + + // OAuth2ConsentScopeName names the scope that authorizes the OAuth2 + // downstream consent page to call the consent flow endpoints. Bound + // resource_id is the authorize-request flow ID. The consent page is + // public (outside /workspace) so it cannot rely on dashboard auth; + // this token is the sole credential binding the browser session to the + // pending authorization request. + OAuth2ConsentScopeName = "oauth2_consent" ) // RoutePattern is one (method, path) pair a Scope grants access to. The path diff --git a/framework/temptoken/sweeper.go b/framework/temptoken/sweeper.go index 57f48896fea..7f48d0e0b8d 100644 --- a/framework/temptoken/sweeper.go +++ b/framework/temptoken/sweeper.go @@ -21,6 +21,7 @@ type SweepWorker struct { sweepInterval time.Duration stopCh chan struct{} stopOnce sync.Once + cancel context.CancelFunc logger schemas.Logger } @@ -44,7 +45,9 @@ func NewSweepWorker(service *Service, logger schemas.Logger) *SweepWorker { // Start begins the sweep loop in a background goroutine. func (w *SweepWorker) Start(ctx context.Context) { - go w.run(ctx) + runCtx, cancel := context.WithCancel(ctx) + w.cancel = cancel + go w.run(runCtx) if w.logger != nil { w.logger.Info("temp-token sweep worker started (interval=%s)", w.sweepInterval) } @@ -54,6 +57,11 @@ func (w *SweepWorker) Start(ctx context.Context) { // panics from redundant shutdown paths. func (w *SweepWorker) Stop() { w.stopOnce.Do(func() { + // Cancel any in-flight sweep so a blocked DB call unwinds promptly, + // then signal run() to exit its ticker loop. + if w.cancel != nil { + w.cancel() + } close(w.stopCh) if w.logger != nil { w.logger.Info("temp-token sweep worker stopped") diff --git a/framework/tracing/store.go b/framework/tracing/store.go index faacb02be81..8fc124fc0b3 100644 --- a/framework/tracing/store.go +++ b/framework/tracing/store.go @@ -420,7 +420,7 @@ func (s *TraceStore) startCleanup() { }() } -// cleanupOldTraces removes traces that have exceeded the TTL +// cleanupOldTraces removes traces and deferred spans that have exceeded the TTL func (s *TraceStore) cleanupOldTraces() { cutoff := time.Now().Add(-s.ttl) count := 0 @@ -436,8 +436,24 @@ func (s *TraceStore) cleanupOldTraces() { return true }) - if count > 0 && s.logger != nil { - s.logger.Debug("tracing: cleaned up %d orphaned traces", count) + // Deferred spans are normally removed by CompleteTrace or ClearDeferredSpan. + // A streaming request that never reaches either (e.g. the final chunk send + // fails without a cancelled context, or the producer goroutine dies before + // the trace completer runs) would otherwise leave its entry — and the + // accumulated response it pins — in the map forever. + deferredCount := 0 + s.deferredSpans.Range(func(key, value any) bool { + info := value.(*DeferredSpanInfo) + if info.StartTime.Before(cutoff) { + if _, ok := s.deferredSpans.LoadAndDelete(key); ok { + deferredCount++ + } + } + return true + }) + + if (count > 0 || deferredCount > 0) && s.logger != nil { + s.logger.Debug("tracing: cleaned up %d orphaned traces and %d orphaned deferred spans", count, deferredCount) } } diff --git a/framework/tracing/store_test.go b/framework/tracing/store_test.go index e9a9d8e9a45..b2fd598ff71 100644 --- a/framework/tracing/store_test.go +++ b/framework/tracing/store_test.go @@ -279,3 +279,35 @@ func TestSetTraceAttribute(t *testing.T) { // Unknown trace ID must be a no-op, not a panic. store.SetTraceAttribute("does-not-exist", "k", "v") } + +func TestCleanupOldTraces_RemovesOrphanedDeferredSpans(t *testing.T) { + store := NewTraceStore(10*time.Millisecond, nil) + defer store.Stop() + + // Simulate a streaming request whose trace completer never ran: the trace + // and its deferred span both exist, and nothing will call CompleteTrace or + // ClearDeferredSpan for them. + orphanTraceID := store.CreateTrace("") + store.StoreDeferredSpan(orphanTraceID, "orphan-span") + + // Let the orphan exceed the TTL, then register a fresh deferred span that + // must survive the sweep. + time.Sleep(20 * time.Millisecond) + freshTraceID := store.CreateTrace("") + store.StoreDeferredSpan(freshTraceID, "fresh-span") + + store.cleanupOldTraces() + + if store.GetDeferredSpan(orphanTraceID) != nil { + t.Error("orphaned deferred span should be removed by TTL cleanup") + } + if store.GetTrace(orphanTraceID) != nil { + t.Error("orphaned trace should be removed by TTL cleanup") + } + if store.GetDeferredSpan(freshTraceID) == nil { + t.Error("fresh deferred span should survive TTL cleanup") + } + if store.GetTrace(freshTraceID) == nil { + t.Error("fresh trace should survive TTL cleanup") + } +} diff --git a/framework/vectorstore/pinecone.go b/framework/vectorstore/pinecone.go index e9d818f7415..2f3035e4b00 100644 --- a/framework/vectorstore/pinecone.go +++ b/framework/vectorstore/pinecone.go @@ -3,6 +3,7 @@ package vectorstore import ( "context" "fmt" + "net" "strings" "sync" @@ -456,13 +457,7 @@ func newPineconeStore(ctx context.Context, config *PineconeConfig, logger schema // Prepare the host URL // For local connections (Pinecone Local), prefix with http:// to disable TLS // See: https://docs.pinecone.io/guides/operations/local-development - host := config.IndexHost.GetValue() - if !strings.HasPrefix(host, "http://") && !strings.HasPrefix(host, "https://") { - // Check if this looks like a local connection - if strings.HasPrefix(host, "localhost") || strings.HasPrefix(host, "127.0.0.1") { - host = "http://" + host - } - } + host := hostWithLocalScheme(config.IndexHost.GetValue()) // Create index connection idxConn, err := client.Index(pinecone.NewIndexConnParams{ Host: host, @@ -485,13 +480,26 @@ func newPineconeStore(ctx context.Context, config *PineconeConfig, logger schema } // getHostWithScheme returns the host with the appropriate scheme. -// For local connections (localhost/127.0.0.1), it adds http:// to disable TLS. +// For local connections (localhost / loopback IPs), it adds http:// to disable TLS. func (s *PineconeStore) getHostWithScheme() string { - host := s.config.IndexHost.GetValue() - if !strings.HasPrefix(host, "http://") && !strings.HasPrefix(host, "https://") { - if strings.HasPrefix(host, "localhost") || strings.HasPrefix(host, "127.0.0.1") { - return "http://" + host - } + return hostWithLocalScheme(s.config.IndexHost.GetValue()) +} + +// hostWithLocalScheme prefixes scheme-less loopback hosts (localhost, 127.0.0.1, +// [::1], with or without port) with http:// so Pinecone Local connections skip TLS. +// See: https://docs.pinecone.io/guides/operations/local-development +func hostWithLocalScheme(host string) string { + if strings.HasPrefix(host, "http://") || strings.HasPrefix(host, "https://") { + return host + } + bare := host + if h, _, err := net.SplitHostPort(bare); err == nil { + bare = h + } + bare = strings.Trim(bare, "[]") + ip := net.ParseIP(bare) + if bare == "localhost" || (ip != nil && ip.IsLoopback()) { + return "http://" + host } return host } diff --git a/framework/vectorstore/pineconehost_test.go b/framework/vectorstore/pineconehost_test.go new file mode 100644 index 00000000000..c1982278655 --- /dev/null +++ b/framework/vectorstore/pineconehost_test.go @@ -0,0 +1,25 @@ +package vectorstore + +import "testing" + +func TestHostWithLocalScheme(t *testing.T) { + tests := []struct { + host string + want string + }{ + {"localhost:5081", "http://localhost:5081"}, + {"localhost", "http://localhost"}, + {"127.0.0.1:5081", "http://127.0.0.1:5081"}, + {"[::1]:5081", "http://[::1]:5081"}, + {"::1", "http://::1"}, + {"http://localhost:5081", "http://localhost:5081"}, // scheme preserved + {"https://index.pinecone.io", "https://index.pinecone.io"}, // scheme preserved + {"index.pinecone.io", "index.pinecone.io"}, // non-local untouched + {"192.168.1.10:5081", "192.168.1.10:5081"}, // private but not loopback + } + for _, tt := range tests { + if got := hostWithLocalScheme(tt.host); got != tt.want { + t.Errorf("hostWithLocalScheme(%q) = %q, want %q", tt.host, got, tt.want) + } + } +} diff --git a/framework/version b/framework/version index 9df886c42a1..428b770e3e2 100644 --- a/framework/version +++ b/framework/version @@ -1 +1 @@ -1.4.2 +1.4.3 diff --git a/helm-charts/bifrost/Chart.yaml b/helm-charts/bifrost/Chart.yaml index f37a6bb6d37..2d15a2ce8f4 100644 --- a/helm-charts/bifrost/Chart.yaml +++ b/helm-charts/bifrost/Chart.yaml @@ -2,7 +2,7 @@ apiVersion: v2 name: bifrost description: A Helm chart for deploying Bifrost - AI Gateway with unified interface for multiple providers type: application -version: 2.1.25 +version: 2.1.26 appVersion: "1.5.12" keywords: - ai diff --git a/helm-charts/bifrost/README.md b/helm-charts/bifrost/README.md index 91546e8d067..841ae25e95c 100644 --- a/helm-charts/bifrost/README.md +++ b/helm-charts/bifrost/README.md @@ -4,10 +4,18 @@ Official Helm charts for deploying [Bifrost](https://github.com/maximhq/bifrost) - a high-performance AI gateway with unified interface for multiple providers. -**Latest Version:** 2.1.25 +**Latest Version:** 2.1.26 ## Changelog +### 2.1.26 + +- Added `bifrost.client.mcpServerAuthMode` (`headers` | `both` | `oauth`) and `bifrost.client.oauth2ServerConfig` (`issuerUrl`, `authCodeTtl`, `accessTokenTtl`, `disableVkIdentity`) to control how `/mcp` authenticates inbound MCP clients. Renders into `client.mcp_server_auth_mode` and `client.oauth2_server_config`. `authCodeTtl` is capped at 900 seconds. +- Added ClickHouse as a `storage.logsStore.type` option. Set `type: clickhouse` and a `storage.logsStore.clickhouse` block (`host` required; optional `port`, `database`, `username`, `password`, `protocol`, `secure`, `dialTimeout`, `cluster`). Renders into `logs_store` with `type: clickhouse`. +- Added the `bedrock_mantle` provider with `bedrock_mantle_key_config` (`region` required; optional `access_key`, `secret_key`, `session_token`, `role_arn`, `external_id`, `session_name`). Added the key-config schema and mutual-exclusion validation in `values.schema.json`. +- Added `toolExecutionTimeout` to `bifrost.mcp.clientConfigs[]` as a per-server override of the global `toolManagerConfig.toolExecutionTimeout`. Accepts a Go duration string (e.g. `"30s"`) or a bare integer treated as seconds. Renders into `mcp.client_configs[].tool_execution_timeout`. +- Added `expires_at` to `bifrost.governance.virtualKeys[]`. Optional RFC3339 timestamp; requests using the virtual key are rejected once it passes. Omit for a key that never expires. + ### 2.1.25 - Added `bifrost.circuitBreakerConfig`. Renders into `circuit_breaker_config` in the generated config JSON. diff --git a/helm-charts/bifrost/templates/_helpers.tpl b/helm-charts/bifrost/templates/_helpers.tpl index d9dd4d76229..3b14298e028 100644 --- a/helm-charts/bifrost/templates/_helpers.tpl +++ b/helm-charts/bifrost/templates/_helpers.tpl @@ -328,6 +328,17 @@ false {{- if .Values.bifrost.client.mcpExternalClientUrl }} {{- $_ := set $client "mcp_external_client_url" .Values.bifrost.client.mcpExternalClientUrl }} {{- end }} +{{- if .Values.bifrost.client.mcpServerAuthMode }} +{{- $_ := set $client "mcp_server_auth_mode" .Values.bifrost.client.mcpServerAuthMode }} +{{- end }} +{{- if .Values.bifrost.client.oauth2ServerConfig }} +{{- $oauth2 := dict }} +{{- with .Values.bifrost.client.oauth2ServerConfig.issuerUrl }}{{- $_ := set $oauth2 "issuer_url" . }}{{- end }} +{{- with .Values.bifrost.client.oauth2ServerConfig.authCodeTtl }}{{- $_ := set $oauth2 "auth_code_ttl" (. | int) }}{{- end }} +{{- with .Values.bifrost.client.oauth2ServerConfig.accessTokenTtl }}{{- $_ := set $oauth2 "access_token_ttl" (. | int) }}{{- end }} +{{- if hasKey .Values.bifrost.client.oauth2ServerConfig "disableVkIdentity" }}{{- $_ := set $oauth2 "disable_vk_identity" .Values.bifrost.client.oauth2ServerConfig.disableVkIdentity }}{{- end }} +{{- if $oauth2 }}{{- $_ := set $client "oauth2_server_config" $oauth2 }}{{- end }} +{{- end }} {{- $_ := set $config "client" $client }} {{- end }} {{- /* Server */ -}} @@ -496,6 +507,7 @@ false {{- if .value }}{{- $_ := set $vk "value" .value }}{{- end }} {{- if .description }}{{- $_ := set $vk "description" .description }}{{- end }} {{- if hasKey . "is_active" }}{{- $_ := set $vk "is_active" .is_active }}{{- end }} +{{- if .expires_at }}{{- $_ := set $vk "expires_at" .expires_at }}{{- end }} {{- if .team_id }}{{- $_ := set $vk "team_id" .team_id }}{{- end }} {{- if .customer_id }}{{- $_ := set $vk "customer_id" .customer_id }}{{- end }} {{- if hasKey . "access_profile_id" }}{{- $_ := set $vk "access_profile_id" .access_profile_id }}{{- end }} @@ -826,6 +838,30 @@ false {{- if $writer }}{{- $_ := set $logsStore "writer" $writer }}{{- end }} {{- end }} {{- $_ := set $config "logs_store" $logsStore }} +{{- else if eq $logsStoreType "clickhouse" }} +{{- if not .Values.storage.logsStore.clickhouse }}{{- fail "ERROR: storage.logsStore.clickhouse is required when storage.logsStore.type is 'clickhouse'." }}{{- end }} +{{- if not .Values.storage.logsStore.clickhouse.host }}{{- fail "ERROR: storage.logsStore.clickhouse.host is required when storage.logsStore.type is 'clickhouse'." }}{{- end }} +{{- $ch := .Values.storage.logsStore.clickhouse }} +{{- $chConfig := dict "host" $ch.host }} +{{- with $ch.port }}{{- $_ := set $chConfig "port" (. | toString) }}{{- end }} +{{- with $ch.database }}{{- $_ := set $chConfig "database" . }}{{- end }} +{{- with $ch.username }}{{- $_ := set $chConfig "username" . }}{{- end }} +{{- with $ch.password }}{{- $_ := set $chConfig "password" . }}{{- end }} +{{- with $ch.protocol }}{{- $_ := set $chConfig "protocol" . }}{{- end }} +{{- if hasKey $ch "secure" }}{{- $_ := set $chConfig "secure" $ch.secure }}{{- end }} +{{- with $ch.dialTimeout }}{{- $_ := set $chConfig "dial_timeout" (. | int) }}{{- end }} +{{- with $ch.cluster }}{{- $_ := set $chConfig "cluster" . }}{{- end }} +{{- $clickhouseLogsStore := dict "enabled" true "type" "clickhouse" "config" $chConfig }} +{{- if .Values.storage.logsStore.writer }} +{{- $writer := dict }} +{{- with .Values.storage.logsStore.writer.maxBatchSize }}{{- $_ := set $writer "max_batch_size" (. | int) }}{{- end }} +{{- with .Values.storage.logsStore.writer.batchInterval }}{{- $_ := set $writer "batch_interval" . }}{{- end }} +{{- with .Values.storage.logsStore.writer.maxBatchBytes }}{{- $_ := set $writer "max_batch_bytes" (. | int) }}{{- end }} +{{- with .Values.storage.logsStore.writer.writeQueueCapacity }}{{- $_ := set $writer "write_queue_capacity" (. | int) }}{{- end }} +{{- with .Values.storage.logsStore.writer.deferredUsageConcurrency }}{{- $_ := set $writer "deferred_usage_concurrency" (. | int) }}{{- end }} +{{- if $writer }}{{- $_ := set $clickhouseLogsStore "writer" $writer }}{{- end }} +{{- end }} +{{- $_ := set $config "logs_store" $clickhouseLogsStore }} {{- else }} {{- $sqliteLogsStore := dict "enabled" true "type" "sqlite" "config" (dict "path" (printf "%s/logs.db" .Values.bifrost.appDir)) }} {{- if .Values.storage.logsStore.writer }} @@ -1081,6 +1117,9 @@ false {{- if $client.toolSyncInterval }} {{- $_ := set $cc "tool_sync_interval" $client.toolSyncInterval }} {{- end }} +{{- if hasKey $client "toolExecutionTimeout" }} +{{- $_ := set $cc "tool_execution_timeout" $client.toolExecutionTimeout }} +{{- end }} {{- if $client.toolPricing }} {{- $_ := set $cc "tool_pricing" $client.toolPricing }} {{- end }} diff --git a/helm-charts/bifrost/values.schema.json b/helm-charts/bifrost/values.schema.json index 2661fdccd23..6005c8084e8 100644 --- a/helm-charts/bifrost/values.schema.json +++ b/helm-charts/bifrost/values.schema.json @@ -547,6 +547,37 @@ "mcpExternalClientUrl": { "type": "string", "description": "Public base URL Bifrost uses as the redirect_uri when acting as an OAuth client to upstream MCP servers. Maps to client.mcp_external_client_url." + }, + "mcpServerAuthMode": { + "type": "string", + "enum": ["headers", "both", "oauth"], + "description": "How /mcp authenticates inbound MCP clients. 'headers' (default): VK/api-key/session headers only, discovery disabled. 'both': accepts header credentials and Bifrost-issued JWTs, discovery enabled. 'oauth': Bifrost JWTs only. Maps to client.mcp_server_auth_mode." + }, + "oauth2ServerConfig": { + "type": "object", + "description": "OAuth2 authorization server settings for /mcp. Only relevant when mcpServerAuthMode is 'both' or 'oauth'. Maps to client.oauth2_server_config.", + "properties": { + "issuerUrl": { + "type": "string", + "description": "Stable public URL advertised as the OAuth2 AS issuer in discovery documents and JWT iss claim. Required for multi-host deployments. Supports env var syntax: \"env.MY_VAR\"." + }, + "authCodeTtl": { + "type": "integer", + "minimum": 1, + "maximum": 900, + "description": "Lifetime of the single-use authorization code in seconds (default: 300, max: 900)." + }, + "accessTokenTtl": { + "type": "integer", + "minimum": 1, + "description": "Lifetime of the issued JWT Bearer token in seconds (default: 600)." + }, + "disableVkIdentity": { + "type": "boolean", + "description": "When true, the OAuth consent flow no longer offers or accepts virtual-key identity. Honored only when an identity provider is configured. Only valid when mcpServerAuthMode is 'oauth'." + } + }, + "additionalProperties": false } }, "additionalProperties": false @@ -868,6 +899,7 @@ "openrouter", "vertex", "cerebras", + "deepseek", "parasail", "perplexity", "sgl", @@ -1658,6 +1690,11 @@ "is_active": { "type": "boolean" }, + "expires_at": { + "type": "string", + "format": "date-time", + "description": "Optional expiry timestamp (RFC3339). Once passed, requests using this virtual key are rejected. Omit for a key that never expires." + }, "team_id": { "type": "string" }, @@ -3499,7 +3536,7 @@ }, "type": { "type": "string", - "enum": ["", "sqlite", "postgres"] + "enum": ["", "sqlite", "postgres", "clickhouse"] }, "maxIdleConns": { "type": "integer", @@ -3514,6 +3551,52 @@ "description": "How often to refresh materialized views. Go duration string (e.g. '30s', '5m', '1h'). Minimum 5s.", "pattern": "^[0-9]+(ns|us|µs|ms|s|m|h)$" }, + "clickhouse": { + "type": "object", + "description": "ClickHouse connection settings (only applies when type is clickhouse)", + "properties": { + "host": { + "type": "string", + "description": "ClickHouse host" + }, + "port": { + "type": ["string", "integer"], + "description": "ClickHouse port. Defaults by protocol: native 9000 (9440 TLS), http 8123 (8443 TLS)" + }, + "database": { + "type": "string", + "description": "ClickHouse database name (default: default)" + }, + "username": { + "type": "string", + "description": "ClickHouse username" + }, + "password": { + "type": "string", + "description": "ClickHouse password (can use env. prefix)" + }, + "protocol": { + "type": "string", + "enum": ["native", "http"], + "description": "ClickHouse wire protocol (default: native)" + }, + "secure": { + "type": "boolean", + "description": "Enable TLS (native: secure=true; http: switches to https)" + }, + "dialTimeout": { + "type": "integer", + "minimum": 1, + "description": "Connection dial timeout in milliseconds (default: 10000)" + }, + "cluster": { + "type": "string", + "description": "Optional cluster name; when set, DDL runs ON CLUSTER with replicated table engines" + } + }, + "required": ["host"], + "additionalProperties": false + }, "writer": { "type": "object", "description": "Async logging writer queue and batch tuning", @@ -4660,6 +4743,41 @@ "required": ["region"], "additionalProperties": false }, + "bedrock_mantle_key_config": { + "type": "object", + "properties": { + "access_key": { + "type": "string", + "description": "AWS access key for SigV4 (can use env. prefix)" + }, + "secret_key": { + "type": "string", + "description": "AWS secret key for SigV4 (can use env. prefix)" + }, + "session_token": { + "type": "string", + "description": "AWS session token for temporary credentials (can use env. prefix)" + }, + "region": { + "type": "string", + "description": "AWS region used to build the bedrock-mantle endpoint host" + }, + "role_arn": { + "type": "string", + "description": "AWS IAM role ARN for AssumeRole (can use env. prefix)" + }, + "external_id": { + "type": "string", + "description": "External ID for AssumeRole (can use env. prefix)" + }, + "session_name": { + "type": "string", + "description": "Role session name for AssumeRole (can use env. prefix)" + } + }, + "required": ["region"], + "additionalProperties": false + }, "vllm_key_config": { "type": "object", "properties": { @@ -4762,6 +4880,9 @@ { "required": ["bedrock_key_config"] }, + { + "required": ["bedrock_mantle_key_config"] + }, { "required": ["vllm_key_config"] } @@ -4778,6 +4899,9 @@ { "required": ["bedrock_key_config"] }, + { + "required": ["bedrock_mantle_key_config"] + }, { "required": ["vllm_key_config"] } @@ -4794,6 +4918,9 @@ { "required": ["bedrock_key_config"] }, + { + "required": ["bedrock_mantle_key_config"] + }, { "required": ["vllm_key_config"] } @@ -4810,6 +4937,28 @@ { "required": ["vertex_key_config"] }, + { + "required": ["bedrock_mantle_key_config"] + }, + { + "required": ["vllm_key_config"] + } + ] + } + }, + { + "required": ["bedrock_mantle_key_config"], + "not": { + "anyOf": [ + { + "required": ["azure_key_config"] + }, + { + "required": ["vertex_key_config"] + }, + { + "required": ["bedrock_key_config"] + }, { "required": ["vllm_key_config"] } @@ -4828,6 +4977,9 @@ }, { "required": ["bedrock_key_config"] + }, + { + "required": ["bedrock_mantle_key_config"] } ] } @@ -5075,6 +5227,10 @@ "type": "string", "description": "Per-client override for tool sync interval" }, + "toolExecutionTimeout": { + "type": ["string", "integer"], + "description": "Per-client override for tool execution timeout. Go duration string (e.g. '30s', '2m') or a bare integer treated as seconds. Overrides the global mcp.toolManagerConfig.toolExecutionTimeout for this server only. Omit or set to 0 to use the global default." + }, "isPingAvailable": { "type": "boolean", "description": "Whether the MCP server supports ping" diff --git a/helm-charts/bifrost/values.yaml b/helm-charts/bifrost/values.yaml index f82e665a468..1271af794ea 100644 --- a/helm-charts/bifrost/values.yaml +++ b/helm-charts/bifrost/values.yaml @@ -200,6 +200,7 @@ bifrost: # Application settings appDir: /app/data port: 8080 + # 0.0.0.0 binds IPv4 interfaces only; use "::" for dual-stack/IPv6-only clusters host: 0.0.0.0 logLevel: info logStyle: json @@ -296,6 +297,14 @@ bifrost: # routingChainMaxDepth: 10 # Maximum depth for routing rule chain evaluation # allowDirectKeys: false # Allow callers to bypass the key pool via x-bf-direct-key + Authorization header # mcpExternalClientUrl: "" # Public base URL used as redirect_uri when Bifrost is an OAuth client to MCP servers + # How /mcp authenticates inbound MCP clients: headers (default), both, or oauth + # mcpServerAuthMode: "headers" + # OAuth2 authorization server settings for /mcp (only used when mcpServerAuthMode is "both" or "oauth") + # oauth2ServerConfig: + # issuerUrl: "" # Stable public issuer URL; required for multi-host deployments. Supports env.VAR_NAME + # authCodeTtl: 300 # Authorization code lifetime in seconds (default 300, max 900) + # accessTokenTtl: 600 # Issued JWT lifetime in seconds (default 600) + # disableVkIdentity: false # Only valid when mcpServerAuthMode is "oauth" # Server configuration server: @@ -401,6 +410,21 @@ bifrost: # region: "us-east-1" # access_key: "env.AWS_ACCESS_KEY_ID" # secret_key: "env.AWS_SECRET_ACCESS_KEY" + # + # # AWS Bedrock Mantle example (requires bedrock_mantle_key_config) + # bedrock_mantle: + # keys: + # - name: "bedrock-mantle-key" + # value: "" + # weight: 1 + # bedrock_mantle_key_config: + # region: "us-east-1" # Required + # access_key: "env.AWS_ACCESS_KEY_ID" + # secret_key: "env.AWS_SECRET_ACCESS_KEY" + # # session_token: "env.AWS_SESSION_TOKEN" + # # role_arn: "" # For AssumeRole + # # external_id: "" + # # session_name: "" # Provider secrets - use existing Kubernetes secrets for provider API keys # These will be injected as environment variables that can be referenced in providers config @@ -435,6 +459,10 @@ bifrost: # - name: "example-https-mcp" # connectionType: "http" # connectionString: "https://my-internal-mcp.corp/mcp" + # # Per-server tool execution timeout override. Go duration string ("30s", "2m") + # # or a bare integer treated as seconds. Overrides toolManagerConfig.toolExecutionTimeout + # # for this server only. Omit or set to 0 to use the global default. + # toolExecutionTimeout: "30s" # # TLS configuration for HTTP and SSE connection types. # # Use when the MCP server presents a self-signed or private CA certificate. # tlsConfig: @@ -717,6 +745,7 @@ bifrost: # description: "Virtual key description" # value: "sk-bf-..." # Optional - auto-generated if omitted # is_active: true + # expires_at: "2026-12-31T23:59:59Z" # Optional RFC3339 expiry; requests rejected once passed. Omit for no expiry # team_id: "team-1" # Mutually exclusive with customer_id # customer_id: "" # Mutually exclusive with team_id # rate_limit_id: "rate-limit-1" @@ -1122,11 +1151,22 @@ storage: logsStore: enabled: true # Backend type for logs store. Empty string uses storage.mode as default - type: "" # Options: sqlite, postgres, or "" (uses storage.mode) + type: "" # Options: sqlite, postgres, clickhouse, or "" (uses storage.mode) # PostgreSQL connection pool tuning (only applies when type is postgres) # maxIdleConns: 5 # maxOpenConns: 50 # matviewRefreshInterval: "30s" # How often to refresh materialized views. Go duration string (e.g. '30s', '5m', '1h'). Minimum 5s. + # ClickHouse connection settings (only applies when type is clickhouse) + # clickhouse: + # host: "clickhouse.default.svc.cluster.local" # Required + # port: "9000" # Defaults by protocol: native 9000 (9440 TLS), http 8123 (8443 TLS) + # database: "default" + # username: "default" + # password: "env.CLICKHOUSE_PASSWORD" + # protocol: "native" # Options: native, http (default: native) + # secure: false # Enable TLS + # dialTimeout: 10000 # Connection dial timeout in milliseconds + # cluster: "" # Optional cluster name; runs DDL ON CLUSTER with replicated engines # Async writer queue and batch tuning. Omitted fields use Bifrost defaults. # writer: # maxBatchSize: 1000 diff --git a/helm-charts/index.yaml b/helm-charts/index.yaml index bd094b9add5..0ada7bddf85 100644 --- a/helm-charts/index.yaml +++ b/helm-charts/index.yaml @@ -1,6 +1,30 @@ apiVersion: v1 entries: bifrost: + - apiVersion: v2 + appVersion: 1.5.12 + created: "2026-07-06T19:36:21.933983+05:30" + description: A Helm chart for deploying Bifrost - AI Gateway with unified interface + for multiple providers + digest: 8484fdfeec2c0f9c7c90024f897ea946d113f4599a20a70d7891465bf1b389eb + home: https://www.getmaxim.ai/bifrost + icon: https://www.getbifrost.ai/favicon.png + keywords: + - ai + - gateway + - llm + - openai + - anthropic + maintainers: + - email: support@getbifrost.ai + name: Bifrost Team + name: bifrost + sources: + - https://github.com/maximhq/bifrost + type: application + urls: + - https://github.com/maximhq/bifrost/releases/download/helm-chart-v2.1.26/bifrost-2.1.26.tgz + version: 2.1.26 - apiVersion: v2 appVersion: 1.5.12 created: "2026-06-25T16:30:36.787778+05:30" @@ -1108,4 +1132,4 @@ entries: urls: - https://maximhq.github.io/bifrost/helm-charts/bifrost-1.3.36.tgz version: 1.3.36 -generated: "2026-06-25T16:30:36.784122+05:30" +generated: "2026-07-06T19:36:21.930494+05:30" diff --git a/plugins/compat/changelog.md b/plugins/compat/changelog.md index e69de29bb2d..bb58d7d9679 100644 --- a/plugins/compat/changelog.md +++ b/plugins/compat/changelog.md @@ -0,0 +1,3 @@ +- feat: drop reasoning when tools are present but `reasoning_with_tool_calls` is unsupported (#4630) +- fix: convert thinking to disabled when tool choice is required for DeepSeek (#4861) +- chore: upgraded core to v1.6.3 and framework to v1.4.3 diff --git a/plugins/compat/conversion.go b/plugins/compat/conversion.go index 7ea01ae03a1..80932405870 100644 --- a/plugins/compat/conversion.go +++ b/plugins/compat/conversion.go @@ -11,7 +11,45 @@ func applyParameterConversion(req *schemas.BifrostRequest) { } if req.ResponsesRequest != nil { flattenNamespaceTools(req.ResponsesRequest) + disableThinkingWithToolChoiceForResponses(req.ResponsesRequest) } + if req.ChatRequest != nil { + disableThinkingWithToolChoice(req.ChatRequest) + } +} + +// disableThinkingWithToolChoice disables thinking when tool_choice forces a tool call. +func disableThinkingWithToolChoice(req *schemas.BifrostChatRequest) { + if req.Provider != schemas.DeepSeek || req.Params == nil || req.Params.ToolChoice == nil { + return + } + tc := req.Params.ToolChoice + if tc.ChatToolChoiceStr != nil && *tc.ChatToolChoiceStr == string(schemas.ChatToolChoiceTypeRequired) { + req.Params.ExtraParams = disableThinking(req.Params.ExtraParams) + } +} + +// disableThinkingWithToolChoiceForResponses disables thinking when tool_choice forces a tool call. +func disableThinkingWithToolChoiceForResponses(req *schemas.BifrostResponsesRequest) { + if req.Provider != schemas.DeepSeek || req.Params == nil || req.Params.ToolChoice == nil { + return + } + tc := req.Params.ToolChoice + if tc.ResponsesToolChoiceStr != nil && *tc.ResponsesToolChoiceStr == string(schemas.ResponsesToolChoiceTypeRequired) { + req.Params.ExtraParams = disableThinking(req.Params.ExtraParams) + } +} + +// disableThinking sets thinking {"type": "disabled"} in extraParams, overwriting +// any caller-provided value. DeepSeek models run with thinking enabled by default, +// and thinking mode rejects forced tool_choice — so a forced tool call requires +// thinking off. +func disableThinking(extraParams map[string]any) map[string]any { + if extraParams == nil { + extraParams = make(map[string]any, 1) + } + extraParams["thinking"] = map[string]any{"type": "disabled"} + return extraParams } // flattenNamespaceTools expands namespace scoped tools into a flat list of tools. diff --git a/plugins/compat/dropparams.go b/plugins/compat/dropparams.go index 7c18e69ff72..9cc28539cba 100644 --- a/plugins/compat/dropparams.go +++ b/plugins/compat/dropparams.go @@ -21,6 +21,7 @@ func dropUnsupportedParams(ctx *schemas.BifrostContext, req *schemas.BifrostRequ if req.ChatRequest != nil && req.ChatRequest.Params != nil { params := req.ChatRequest.Params + hasSupportedTools := len(params.Tools) > 0 && isSupported["tools"] if params.Audio != nil && !isSupported["audio"] { params.Audio = nil @@ -68,9 +69,13 @@ func dropUnsupportedParams(ctx *schemas.BifrostContext, req *schemas.BifrostRequ params.PromptCacheRetention = nil dropped = append(dropped, "prompt_cache_retention") } - if params.Reasoning != nil && !isSupported["reasoning"] { - params.Reasoning = nil - dropped = append(dropped, "reasoning") + if params.Reasoning != nil { + // for chat completions, some models do not support reasoning_effort + // with tools + if !isSupported["reasoning"] || (hasSupportedTools && !isSupported["reasoning_with_tool_calls"]) { + params.Reasoning = nil + dropped = append(dropped, "reasoning") + } } if params.ResponseFormat != nil && !isSupported["response_format"] { params.ResponseFormat = nil diff --git a/plugins/compat/dropparams_test.go b/plugins/compat/dropparams_test.go index 748d1556e6f..de69bb64e20 100644 --- a/plugins/compat/dropparams_test.go +++ b/plugins/compat/dropparams_test.go @@ -170,3 +170,52 @@ func TestDropUnsupportedParams_ChatMaxCompletionTokensUnchanged(t *testing.T) { t.Errorf("max_completion_tokens = preserved, want dropped (no token cap supported)") } } + +func TestDropUnsupportedParams_ChatReasoningWithUnsupportedTools(t *testing.T) { + newChat := func() *schemas.BifrostRequest { + return &schemas.BifrostRequest{ + RequestType: schemas.ChatCompletionRequest, + ChatRequest: &schemas.BifrostChatRequest{ + Provider: schemas.OpenAI, + Model: "reasoning-no-tools-model", + Params: &schemas.ChatParameters{ + Reasoning: &schemas.ChatReasoning{}, + Tools: []schemas.ChatTool{{ + Type: schemas.ChatToolTypeFunction, + Function: &schemas.ChatToolFunction{ + Name: "get_weather", + Description: schemas.Ptr("Returns weather"), + }, + }}, + }, + }, + } + } + + preserveReasoning := newChat() + dropped := dropUnsupportedParams(newTestContext(), preserveReasoning, []string{"reasoning"}) + if preserveReasoning.ChatRequest.Params.Reasoning == nil { + t.Fatalf("reasoning = dropped, want preserved when tools are also dropped") + } + if preserveReasoning.ChatRequest.Params.Tools != nil { + t.Fatalf("tools = preserved, want dropped") + } + if slices.Contains(dropped, "reasoning") { + t.Errorf("reasoning reported in dropped=%v, want absent", dropped) + } + if !slices.Contains(dropped, "tools") { + t.Errorf("tools not reported in dropped=%v, want present", dropped) + } + + dropReasoning := newChat() + dropped = dropUnsupportedParams(newTestContext(), dropReasoning, []string{"reasoning", "tools"}) + if dropReasoning.ChatRequest.Params.Reasoning != nil { + t.Fatalf("reasoning = preserved, want dropped when tools survive without reasoning_with_tool_calls") + } + if dropReasoning.ChatRequest.Params.Tools == nil { + t.Fatalf("tools = dropped, want preserved") + } + if !slices.Contains(dropped, "reasoning") { + t.Errorf("reasoning not reported in dropped=%v, want present", dropped) + } +} diff --git a/plugins/compat/version b/plugins/compat/version index 5a48b6be2af..0e7400f186e 100644 --- a/plugins/compat/version +++ b/plugins/compat/version @@ -1 +1 @@ -0.1.24 +0.1.25 diff --git a/plugins/governance/changelog.md b/plugins/governance/changelog.md index e69de29bb2d..68cf502adb5 100644 --- a/plugins/governance/changelog.md +++ b/plugins/governance/changelog.md @@ -0,0 +1,8 @@ +- feat: added expiry enforcement for virtual keys (#4887) +- feat: complexity analyzer stemming support alongside exact keyword match and a no-signal fallback (#4708, #4791) +- feat: added `virtualKeysByID` secondary index with cached signing key and VK lookups on the `/mcp` JWT auth path (#4783) +- feat: virtual key values use `schemas.SecretVar` to support the env store (#4817) +- fix: skip O(N) reference refresh on request-time rate-limit and budget reset (#4883, closes #4851) +- fix: skip model check for Responses lifecycle APIs (#4920) +- fix: empty tool call result insertion failures (#4925) +- chore: upgraded core to v1.6.3 and framework to v1.4.3 diff --git a/plugins/governance/complexity/analyzer.go b/plugins/governance/complexity/analyzer.go index c10e5deb88f..dcdc0b20cc7 100644 --- a/plugins/governance/complexity/analyzer.go +++ b/plugins/governance/complexity/analyzer.go @@ -31,58 +31,51 @@ func NewComplexityAnalyzerWithConfig(config *AnalyzerConfig) *ComplexityAnalyzer // Analyze computes complexity scores from the normalized input. func (a *ComplexityAnalyzer) Analyze(input ComplexityInput) *ComplexityResult { - // Select scan mask based on whether conversation history is present. - lastScanMask := lastTextBaseScanMask + // Extract lexical signals from last user message and system prompt. + lastSignals := a.matcher.analyzeText(input.LastUserText, lastTextBaseScanMask) + wordCount := lastSignals.wordCount + hasPositiveSignal := hasPositiveSignal(lastSignals) + hasSimpleSignal := lastSignals.simpleCount > 0 + + var convScore float64 if len(input.PriorUserTexts) > 0 { - lastScanMask = lastTextFullScanMask + convScore = a.scoreConversationContext(input.PriorUserTexts) + } + isContinuation := isContinuationFollowup(lastSignals, convScore) + if !hasPositiveSignal && !hasSimpleSignal && !isContinuation { + return nil } - // Extract lexical signals from last user message and system prompt. - lastSignals := a.matcher.analyzeText(input.LastUserText, lastScanMask) - systemSignals := a.matcher.analyzeText(input.SystemText, systemTextScanMask) + systemSignals := textSignalCounts{} + if hasPositiveSignal { + systemSignals = a.matcher.analyzeText(input.SystemText, systemTextScanMask) + } // Score primary message signals. userCodeScore := scoreCount(lastSignals.codeCount, 3) reasoningScore := scoreCount(lastSignals.reasoningCount, 2) userTechnicalScore := scoreCount(lastSignals.technicalCount, 3) userSimpleScore := scoreCount(lastSignals.simpleCount, 2) - outputScore := scoreOutputComplexity(lastSignals) - tokenScore := scoreTokenCount(lastSignals.wordCount) + tokenScore := 0.0 + if hasPositiveSignal || isContinuation { + tokenScore = scoreTokenCount(wordCount) + } - // System prompt provides soft lexical context for code/technical/simple signals, - // but never drives reasoning override, token count, or output complexity. + // System prompt provides soft lexical context for code/technical signals, + // but never drives reasoning override or token count. systemCodeScore := scoreCount(systemSignals.codeCount, 3) systemTechnicalScore := scoreCount(systemSignals.technicalCount, 3) - systemSimpleScore := scoreCount(systemSignals.simpleCount, 2) codeScore := clamp(userCodeScore+(systemCodeScore*systemPromptAssistFactor), 0.0, 1.0) technicalScore := clamp(userTechnicalScore+(systemTechnicalScore*systemPromptAssistFactor), 0.0, 1.0) - simpleScore := clamp(userSimpleScore+(systemSimpleScore*systemPromptAssistFactor), 0.0, 1.0) - - // Conditional simple dampener: only apply full dampener on short, low-signal asks. - wordCount := lastSignals.wordCount - effectiveSimpleWeight := simpleWeight - signalCount := 0 - if userCodeScore >= 0.3 { - signalCount++ - } - if userTechnicalScore >= 0.3 { - signalCount++ - } - if reasoningScore >= 0.3 { - signalCount++ - } - if lastSignals.simpleCount > 0 && (wordCount >= 30 || signalCount >= 2) { - effectiveSimpleWeight = 0.01 - } codeContribution := codeScore * codeWeight reasoningContribution := reasoningScore * reasoningWeight technicalContribution := technicalScore * technicalWeight - simplePenalty := -(simpleScore * effectiveSimpleWeight) + simplePenalty := -(userSimpleScore * simpleWeight) tokenContribution := tokenScore * tokenCountWeight - // Weighted sum for last message (output complexity applied separately as a score floor). + // Weighted sum for last message. lastMsgScore := codeContribution + reasoningContribution + technicalContribution + @@ -92,12 +85,10 @@ func (a *ComplexityAnalyzer) Analyze(input ComplexityInput) *ComplexityResult { // Conversation context blending (prior user turns only). var blended float64 - var convScore float64 - if len(input.PriorUserTexts) > 0 { - convScore = a.scoreConversationContext(input.PriorUserTexts) + if len(input.PriorUserTexts) > 0 && (hasPositiveSignal || isContinuation) { lastWeight := defaultLastMessageBlendWeight contextWeight := defaultConversationBlendWeight - if isReferentialFollowup(lastSignals, lastMsgScore, convScore, wordCount) { + if isContinuation { lastWeight = referentialLastMessageBlendWeight contextWeight = referentialConversationBlendWeight } @@ -108,15 +99,6 @@ func (a *ComplexityAnalyzer) Analyze(input ComplexityInput) *ComplexityResult { blended = lastMsgScore } - // Output complexity as a score floor: strong output signals set a minimum score. - outputFloorMinScore := 0.0 - if outputScore > 0.5 { - outputFloorMinScore = outputScore * 0.5 - if blended < outputFloorMinScore { - blended = outputFloorMinScore - } - } - finalScore := clamp(blended, 0.0, 1.0) // Tier classification with reasoning override. @@ -170,23 +152,15 @@ func (a *ComplexityAnalyzer) scoreConversationContext(priorUserTexts []string) f return math.Min(1.0, weightedTotal/totalWeight) } -func isReferentialFollowup(signals textSignalCounts, lastMsgScore, convScore float64, wordCount int) bool { - if wordCount == 0 || wordCount > referentialMaxWordCount { - return false - } - if lastMsgScore >= referentialMaxStandaloneScore || convScore < referentialMinContextScore { - return false - } - if signals.taskShiftCount > 0 { +func hasPositiveSignal(signals textSignalCounts) bool { + return signals.codeCount > 0 || signals.reasoningCount > 0 || signals.technicalCount > 0 +} + +func isContinuationFollowup(signals textSignalCounts, convScore float64) bool { + if convScore < referentialMinContextScore { return false } - if signals.referentialPhraseCount > 0 { - return true - } - - hasReference := signals.referentialReferenceCount > 0 - hasAction := signals.referentialActionCount > 0 - return hasReference && hasAction + return signals.continuationPhraseCount > 0 } func (a *ComplexityAnalyzer) classifyTier(score float64) string { diff --git a/plugins/governance/complexity/analyzer_test.go b/plugins/governance/complexity/analyzer_test.go index 7fea167c90c..680a9f77f31 100644 --- a/plugins/governance/complexity/analyzer_test.go +++ b/plugins/governance/complexity/analyzer_test.go @@ -59,6 +59,31 @@ func TestAnalyze_Hello(t *testing.T) { if result.Tier != "SIMPLE" { t.Errorf("expected SIMPLE tier for greeting, got %s (score=%.3f)", result.Tier, result.Score) } + if result.Score != 0.0 { + t.Errorf("expected simple-only greeting to clamp to 0.0, got %.3f", result.Score) + } +} + +func TestAnalyze_NoSignalFallsBackButSimpleSignalClassifies(t *testing.T) { + a := NewComplexityAnalyzer() + + noSignal := a.Analyze(ComplexityInput{ + LastUserText: "2+3", + }) + if noSignal != nil { + t.Fatalf("expected no-signal arithmetic prompt to be unclassified, got %s (score=%.3f)", noSignal.Tier, noSignal.Score) + } + + simpleSignal := a.Analyze(ComplexityInput{ + LastUserText: "translate this to spanish", + }) + if simpleSignal == nil { + t.Fatalf("expected simple keyword prompt to classify") + } + if simpleSignal.Tier != TierSimple || simpleSignal.Score != 0.0 { + t.Fatalf("expected simple keyword prompt to classify as SIMPLE with 0.0 score, got %s (score=%.3f)", + simpleSignal.Tier, simpleSignal.Score) + } } func TestAnalyze_CodeRequest(t *testing.T) { @@ -77,11 +102,11 @@ func TestAnalyze_Complex(t *testing.T) { a := NewComplexityAnalyzer() result := a.Analyze(ComplexityInput{ - LastUserText: "Design a distributed authentication system using Kubernetes with encryption and load balancer", + LastUserText: "Design a distributed authentication architecture using Kubernetes, encryption, load balancer, failover, RBAC, OIDC, audit log, and connection pool idempotency.", }) - if result.Tier != "COMPLEX" && result.Tier != "REASONING" { - t.Errorf("expected COMPLEX or REASONING tier for architecture request, got %s (score=%.3f)", result.Tier, result.Score) + if result.Tier == "SIMPLE" { + t.Errorf("expected MEDIUM or higher tier for architecture request, got %s (score=%.3f)", result.Tier, result.Score) } } @@ -97,27 +122,28 @@ func TestAnalyze_Reasoning(t *testing.T) { } } -func TestAnalyze_OutputComplexity(t *testing.T) { +func TestAnalyze_OutputComplexityRequiresVisibleSignal(t *testing.T) { a := NewComplexityAnalyzer() result := a.Analyze(ComplexityInput{ LastUserText: "List every AWS service and explain each one with examples", }) - if result.Tier == "SIMPLE" { - t.Errorf("expected non-SIMPLE tier for output-heavy request, got %s (score=%.3f)", result.Tier, result.Score) + if result != nil { + t.Errorf("expected output-heavy request without visible signals to be unclassified, got %s (score=%.3f)", result.Tier, result.Score) } } -func TestAnalyze_ConversationContext(t *testing.T) { +func TestAnalyze_ConversationContextDoesNotClassifyNoSignalLatestTurn(t *testing.T) { a := NewComplexityAnalyzer() - // Short follow-up with no context stays SIMPLE. noCtx := a.Analyze(ComplexityInput{ LastUserText: "Why?", }) + if noCtx != nil { + t.Fatalf("expected no-signal latest turn without context to be unclassified, got %s (score=%.3f)", noCtx.Tier, noCtx.Score) + } - // Same follow-up with technical conversation history gets a higher score. withCtx := a.Analyze(ComplexityInput{ LastUserText: "Why?", PriorUserTexts: []string{ @@ -127,9 +153,8 @@ func TestAnalyze_ConversationContext(t *testing.T) { }, }) - if withCtx.Score <= noCtx.Score { - t.Errorf("expected conversation context to raise score: noCtx=%.3f, withCtx=%.3f", - noCtx.Score, withCtx.Score) + if withCtx != nil { + t.Errorf("expected complex history not to classify a no-signal latest turn, got %s (score=%.3f)", withCtx.Tier, withCtx.Score) } } @@ -155,7 +180,7 @@ func TestAnalyze_ConversationContextDoesNotDiluteStrongLastMessage(t *testing.T) } } -func TestAnalyze_ReferentialFollowupLiftsShortTechnicalContinuation(t *testing.T) { +func TestAnalyze_ContinuationPhraseLiftsTechnicalContinuation(t *testing.T) { a := NewComplexityAnalyzer() result := a.Analyze(ComplexityInput{ @@ -167,27 +192,30 @@ func TestAnalyze_ReferentialFollowupLiftsShortTechnicalContinuation(t *testing.T }, }) + if result == nil { + t.Fatalf("expected continuation phrase with prior context to classify") + } if result.Tier == "SIMPLE" { - t.Fatalf("expected short referential follow-up to lift above SIMPLE, got %s (score=%.3f)", result.Tier, result.Score) + t.Fatalf("expected continuation to lift above SIMPLE, got %s (score=%.3f)", result.Tier, result.Score) } if result.Score < simpleMediumBoundary { t.Fatalf("expected score above SIMPLE threshold, got %.3f", result.Score) } } -func TestAnalyze_ReferentialFollowupRequiresRealContext(t *testing.T) { +func TestAnalyze_ContinuationPhraseRequiresRealContext(t *testing.T) { a := NewComplexityAnalyzer() result := a.Analyze(ComplexityInput{ LastUserText: "do it", }) - if result.Tier != "SIMPLE" { - t.Fatalf("expected SIMPLE tier without prior context, got %s (score=%.3f)", result.Tier, result.Score) + if result != nil { + t.Fatalf("expected continuation phrase without prior context to be unclassified, got %s (score=%.3f)", result.Tier, result.Score) } } -func TestAnalyze_TaskShiftFollowupDoesNotUseReferentialLift(t *testing.T) { +func TestAnalyze_SimpleKeywordFollowupDoesNotUseContext(t *testing.T) { a := NewComplexityAnalyzer() result := a.Analyze(ComplexityInput{ @@ -198,12 +226,15 @@ func TestAnalyze_TaskShiftFollowupDoesNotUseReferentialLift(t *testing.T) { }, }) + if result == nil { + t.Fatalf("expected simple keyword follow-up to classify") + } if result.Score >= mediumComplexBoundary { - t.Fatalf("expected task-shift request to stay below COMPLEX threshold, got %.3f", result.Score) + t.Fatalf("expected simple keyword follow-up to stay below COMPLEX threshold, got %.3f", result.Score) } } -func TestAnalyze_LimitingTaskShiftDoesNotUseReferentialLift(t *testing.T) { +func TestAnalyze_UnmatchedFollowupDoesNotUseContext(t *testing.T) { a := NewComplexityAnalyzer() result := a.Analyze(ComplexityInput{ @@ -214,8 +245,8 @@ func TestAnalyze_LimitingTaskShiftDoesNotUseReferentialLift(t *testing.T) { }, }) - if result.Score >= mediumComplexBoundary { - t.Fatalf("expected limiting summary request to stay below COMPLEX threshold, got %.3f", result.Score) + if result != nil { + t.Fatalf("expected unmatched follow-up not to use context, got %s (score=%.3f)", result.Tier, result.Score) } } @@ -226,8 +257,8 @@ func TestAnalyze_RecentContextOutweighsOlderContext(t *testing.T) { LastUserText: "do it", PriorUserTexts: []string{ "Hello there.", - "Thanks.", "Design a distributed authentication system with RBAC, OIDC, and regional failover.", + "Debug the API gateway encryption middleware and Kubernetes connection pool behavior.", }, }) @@ -235,11 +266,14 @@ func TestAnalyze_RecentContextOutweighsOlderContext(t *testing.T) { LastUserText: "do it", PriorUserTexts: []string{ "Design a distributed authentication system with RBAC, OIDC, and regional failover.", - "Hello there.", + "Debug the API gateway encryption middleware and Kubernetes connection pool behavior.", "Thanks.", }, }) + if recentTech == nil || olderTech == nil { + t.Fatalf("expected both continuation cases to classify, got recent=%v older=%v", recentTech, olderTech) + } if recentTech.Score <= olderTech.Score { t.Fatalf("expected more recent technical context to matter more: recent=%.3f older=%.3f", recentTech.Score, olderTech.Score) @@ -264,25 +298,25 @@ func TestAnalyze_SystemPromptBoost(t *testing.T) { } } -func TestAnalyze_SystemPromptDampener(t *testing.T) { +func TestAnalyze_SystemPromptSimpleSignalsIgnored(t *testing.T) { a := NewComplexityAnalyzer() base := a.Analyze(ComplexityInput{ - LastUserText: "Explain how databases work", + LastUserText: "Explain how database code works", }) - dampened := a.Analyze(ComplexityInput{ - LastUserText: "Explain how databases work", + withSimpleSystemPrompt := a.Analyze(ComplexityInput{ + LastUserText: "Explain how database code works", SystemText: "You are a beginner tutor. Keep answers simple, brief, and concise.", }) - if dampened.Score >= base.Score { - t.Errorf("expected system prompt to dampen score: base=%.3f, dampened=%.3f", - base.Score, dampened.Score) + if withSimpleSystemPrompt.Score != base.Score { + t.Errorf("expected simple system-prompt terms to be ignored: base=%.3f, withSimpleSystemPrompt=%.3f", + base.Score, withSimpleSystemPrompt.Score) } } -func TestAnalyze_SystemPromptLexicalAssistDoesNotOverPromoteSimpleWebhook(t *testing.T) { +func TestAnalyze_SystemPromptLexicalAssistDoesNotOverPromoteSimpleCodeDefinition(t *testing.T) { a := NewComplexityAnalyzer() result := a.Analyze(ComplexityInput{ @@ -301,32 +335,26 @@ func TestAnalyze_EmptyInput(t *testing.T) { result := a.Analyze(ComplexityInput{}) - if result.Tier != "SIMPLE" { - t.Errorf("expected SIMPLE tier for empty input, got %s", result.Tier) - } - if result.Score != 0.0 { - t.Errorf("expected 0.0 score for empty input, got %.3f", result.Score) + if result != nil { + t.Errorf("expected empty input to be unclassified, got %s (score=%.3f)", result.Tier, result.Score) } } func TestAnalyze_ReasoningOverrideNotTooEager(t *testing.T) { a := NewComplexityAnalyzer() - // Two weak reasoning markers should NOT force REASONING result := a.Analyze(ComplexityInput{ LastUserText: "Why does React re-render, and what if I use useMemo?", }) - if result.Tier == "REASONING" { - t.Errorf("expected non-REASONING tier for casual question with weak markers, got %s (score=%.3f)", - result.Tier, result.Score) + if result != nil { + t.Errorf("expected removed broad reasoning markers to be unclassified, got %s (score=%.3f)", result.Tier, result.Score) } } -func TestAnalyze_SimpleDampenerConditional(t *testing.T) { +func TestAnalyze_SimpleKeywordDoesNotSuppressTechnicalSignals(t *testing.T) { a := NewComplexityAnalyzer() - // "What is" + technical term should not be over-dampened result := a.Analyze(ComplexityInput{ LastUserText: "What is eventual consistency in distributed systems with sharding?", }) @@ -371,8 +399,8 @@ func TestAnalyze_MultiTenantSSOArchitecture(t *testing.T) { LastUserText: "Design a multi-tenant authentication service for a SaaS platform on Kubernetes. Requirements: RBAC with custom roles per tenant, audit logging for all auth events, regional failover across two AWS regions, and support for both SAML 2.0 and OIDC enterprise SSO. Include the data model and the request flow for a login.", }) - if result.Tier != "COMPLEX" && result.Tier != "REASONING" { - t.Errorf("expected COMPLEX or REASONING tier for multi-tenant SSO architecture prompt, got %s (score=%.3f)", + if result.Tier == "SIMPLE" { + t.Errorf("expected MEDIUM or higher tier for multi-tenant SSO architecture prompt, got %s (score=%.3f)", result.Tier, result.Score) } } @@ -489,50 +517,40 @@ func TestAnalyze_VectorDatabaseTradeoffRecommendation(t *testing.T) { } } -func TestIsReferentialFollowup_GuardBranches(t *testing.T) { +func TestIsContinuationFollowup_GuardBranches(t *testing.T) { tests := []struct { - name string - lastText string - lastMsgScore float64 - convScore float64 - wordCount int - expected bool + name string + lastText string + convScore float64 + expected bool }{ - {"phrase_match_ok", "do it", 0.05, 0.30, 2, true}, - {"phrase_match_at_word_cap", "do it now please right away", 0.05, 0.30, 6, true}, - {"phrase_match_over_word_cap", "do it now please right away ok", 0.05, 0.30, 7, false}, - {"phrase_match_zero_words", "", 0.0, 0.30, 0, false}, - {"phrase_match_score_at_threshold", "do it", 0.15, 0.30, 2, false}, - {"phrase_match_score_just_below_threshold", "do it", 0.149, 0.30, 2, true}, - {"phrase_match_conv_just_below_threshold", "do it", 0.05, 0.199, 2, false}, - {"phrase_match_conv_at_threshold", "do it", 0.05, 0.20, 2, true}, - {"task_shift_blocks_phrase_match", "translate it", 0.05, 0.30, 2, false}, - {"task_shift_blocks_summarize", "summarize it", 0.05, 0.30, 2, false}, - {"task_shift_one_sentence_blocks", "rewrite it in one sentence", 0.05, 0.30, 5, false}, - {"multi_signal_fix_it", "fix it", 0.05, 0.30, 2, true}, - {"multi_signal_make_it_shorter", "make it shorter", 0.05, 0.30, 3, true}, - {"multi_signal_rewrite_it", "rewrite it", 0.05, 0.30, 2, true}, - {"multi_signal_use_that", "use that", 0.05, 0.30, 2, true}, - {"multi_signal_answer_previous", "answer the previous question", 0.05, 0.30, 4, true}, - {"action_only_no_deictic", "fix the race condition", 0.05, 0.30, 4, false}, - {"deictic_only_no_action", "this is great", 0.05, 0.30, 3, false}, - {"unrelated_short_text", "hello there friend", 0.05, 0.30, 3, false}, + {"phrase_match_ok", "do it", 0.30, true}, + {"phrase_match_longer_text", "do it now please right away ok", 0.30, true}, + {"no_phrase", "", 0.30, false}, + {"phrase_match_conv_just_below_threshold", "do it", 0.199, false}, + {"phrase_match_conv_at_threshold", "do it", 0.20, true}, + {"explicit_use_option", "use option 2", 0.30, true}, + {"retry_is_code_not_continuation", "retry", 0.30, false}, + {"former_inferred_fix_it", "fix it", 0.30, false}, + {"former_inferred_make_it_shorter", "make it shorter", 0.30, false}, + {"former_inferred_answer_previous", "answer the previous question", 0.30, false}, + {"unrelated_short_text", "hello there friend", 0.30, false}, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { matcher := newCompiledKeywordMatcher(defaultFullKeywordConfig()) - signals := matcher.analyzeText(tt.lastText, lastTextFullScanMask) - got := isReferentialFollowup(signals, tt.lastMsgScore, tt.convScore, tt.wordCount) + signals := matcher.analyzeText(tt.lastText, lastTextBaseScanMask) + got := isContinuationFollowup(signals, tt.convScore) if got != tt.expected { - t.Errorf("isReferentialFollowup(%q, last=%.3f, conv=%.3f, words=%d) = %v, want %v", - tt.lastText, tt.lastMsgScore, tt.convScore, tt.wordCount, got, tt.expected) + t.Errorf("isContinuationFollowup(%q, conv=%.3f) = %v, want %v", + tt.lastText, tt.convScore, got, tt.expected) } }) } } -func TestAnalyze_ReferentialMultiSignalDetection(t *testing.T) { +func TestAnalyze_ExplicitContinuationPhrasesUseContext(t *testing.T) { a := NewComplexityAnalyzer() techPriors := []string{ @@ -545,10 +563,11 @@ func TestAnalyze_ReferentialMultiSignalDetection(t *testing.T) { name string lastText string }{ - {"fix_it", "fix it"}, - {"make_it_shorter", "make it shorter"}, - {"rewrite_it", "rewrite it"}, - {"do_this", "do this"}, + {"do_it", "do it"}, + {"try_again", "try again"}, + {"go_ahead", "go ahead"}, + {"same_thing", "same thing"}, + {"use_option", "use option 2"}, } for _, tt := range tests { @@ -565,7 +584,7 @@ func TestAnalyze_ReferentialMultiSignalDetection(t *testing.T) { } } -func TestAnalyze_ReferentialPhraseDoesNotHijackStrongAsk(t *testing.T) { +func TestAnalyze_ContinuationPhraseDoesNotHijackStrongAsk(t *testing.T) { a := NewComplexityAnalyzer() result := a.Analyze(ComplexityInput{ @@ -576,7 +595,7 @@ func TestAnalyze_ReferentialPhraseDoesNotHijackStrongAsk(t *testing.T) { }) if result.Tier == "SIMPLE" { - t.Fatalf("expected high-signal message to stay above SIMPLE despite referential phrase, got %s (score=%.3f)", + t.Fatalf("expected high-signal message to stay above SIMPLE despite continuation phrase, got %s (score=%.3f)", result.Tier, result.Score) } } @@ -594,6 +613,7 @@ func TestAnalyze_RegressionAnchors(t *testing.T) { name string lastText string priors []string + expectNil bool minTier string // tier must be at least this rank (or empty for "any") maxTier string // tier must be at most this rank (or empty for "any") mustNotEqualTiers []string @@ -617,16 +637,16 @@ func TestAnalyze_RegressionAnchors(t *testing.T) { maxTier: "MEDIUM", }, { - name: "summarize_after_tech_thread_stays_simple", - lastText: "summarize it in one sentence", - priors: techPriors, - maxTier: "MEDIUM", + name: "summarize_after_tech_thread_is_unclassified", + lastText: "summarize it in one sentence", + priors: techPriors, + expectNil: true, }, { - name: "do_it_with_empty_priors_stays_simple", - lastText: "do it", - priors: nil, - maxTier: "SIMPLE", + name: "do_it_with_empty_priors_is_unclassified", + lastText: "do it", + priors: nil, + expectNil: true, }, { name: "strong_arch_ask_with_smalltalk_priors_stays_strong", @@ -651,6 +671,15 @@ func TestAnalyze_RegressionAnchors(t *testing.T) { PriorUserTexts: tt.priors, }) + if tt.expectNil { + if result != nil { + t.Fatalf("expected unclassified result, got tier=%s score=%.3f", result.Tier, result.Score) + } + return + } + if result == nil { + t.Fatalf("expected classified result") + } if tt.minTier != "" && tierRank[result.Tier] < tierRank[tt.minTier] { t.Errorf("tier=%s, expected at least %s (score=%.3f)", result.Tier, tt.minTier, result.Score) } @@ -765,7 +794,8 @@ func TestKeywordMatchModeFor(t *testing.T) { } func TestBuildWordPresenceSet_UnicodeWords(t *testing.T) { - words := buildWordPresenceSet("la sécurité du réseau protège les données") + text := "la sécurité du réseau protège les données" + words := buildWordPresenceSet(text, countWordsNoAlloc(text)) if _, ok := words["sécurité"]; !ok { t.Fatalf("expected unicode word to be preserved in presence set") diff --git a/plugins/governance/complexity/config.go b/plugins/governance/complexity/config.go index 57bda9eb77c..4b42d03a7fb 100644 --- a/plugins/governance/complexity/config.go +++ b/plugins/governance/complexity/config.go @@ -42,19 +42,11 @@ type AnalyzerConfig = configstore.ComplexityAnalyzerConfig // KeywordConfig is the full internal keyword set used by the compiled matcher. type KeywordConfig struct { - CodeKeywords []string - StrongReasoningKeywords []string - WeakReasoningKeywords []string - TechnicalKeywords []string - SimpleKeywords []string - EnumTriggers []string - ComprehensivenessMarkers []string - ElaborationMarkers []string - LimitingQualifiers []string - ReferentialPhrases []string - ReferentialReferenceWords []string - ReferentialActionWords []string - TaskShiftPhrases []string + CodeKeywords []string + StrongReasoningKeywords []string + TechnicalKeywords []string + SimpleKeywords []string + ContinuationPhrases []string } // DefaultTierBoundaries returns the built-in classification thresholds. @@ -116,19 +108,11 @@ func mergeEditableKeywordsOntoDefaults(editable EditableKeywordConfig) KeywordCo func defaultFullKeywordConfig() KeywordConfig { return KeywordConfig{ - CodeKeywords: cloneStringSlice(codeKeywords), - StrongReasoningKeywords: cloneStringSlice(strongReasoningKeywords), - WeakReasoningKeywords: cloneStringSlice(weakReasoningKeywords), - TechnicalKeywords: cloneStringSlice(technicalKeywords), - SimpleKeywords: cloneStringSlice(simpleKeywords), - EnumTriggers: cloneStringSlice(enumTriggers), - ComprehensivenessMarkers: cloneStringSlice(comprehensivenessMarkers), - ElaborationMarkers: cloneStringSlice(elaborationMarkers), - LimitingQualifiers: cloneStringSlice(limitingQualifiers), - ReferentialPhrases: cloneStringSlice(referentialPhrases), - ReferentialReferenceWords: cloneStringSlice(referentialReferenceWords), - ReferentialActionWords: cloneStringSlice(referentialActionWords), - TaskShiftPhrases: cloneStringSlice(taskShiftPhrases), + CodeKeywords: cloneStringSlice(codeKeywords), + StrongReasoningKeywords: cloneStringSlice(strongReasoningKeywords), + TechnicalKeywords: cloneStringSlice(technicalKeywords), + SimpleKeywords: cloneStringSlice(simpleKeywords), + ContinuationPhrases: cloneStringSlice(continuationPhrases), } } diff --git a/plugins/governance/complexity/keywords.go b/plugins/governance/complexity/keywords.go index 3f5bba5b0e5..d0ab46e38c3 100644 --- a/plugins/governance/complexity/keywords.go +++ b/plugins/governance/complexity/keywords.go @@ -6,18 +6,15 @@ const ( codeWeight = 0.30 reasoningWeight = 0.25 technicalWeight = 0.25 - simpleWeight = 0.05 // dampener, subtracted + simpleWeight = 0.05 tokenCountWeight = 0.10 systemPromptAssistFactor = 0.25 defaultLastMessageBlendWeight = 0.60 defaultConversationBlendWeight = 0.40 referentialLastMessageBlendWeight = 0.35 referentialConversationBlendWeight = 0.65 - referentialMaxStandaloneScore = 0.15 - referentialMaxWordCount = 6 referentialMinContextScore = 0.20 wordPresenceSetMinBytes = 8 * 1024 - // Output complexity is applied as a score floor, not a weighted dimension ) // --- Keyword lists --- @@ -36,7 +33,6 @@ var codeKeywords = []string{ "cel", "auto-routing", "rwmutex", "goroutine", } -// Reasoning markers, split into strong and weak for override logic. var strongReasoningKeywords = []string{ "step by step", "think through", "tradeoffs", "pros and cons", "justify", "critique", "implications", "explain why", @@ -45,14 +41,6 @@ var strongReasoningKeywords = []string{ "explain your reasoning", "weigh the tradeoffs", "recommend a design", } -var weakReasoningKeywords = []string{ - "reason", "analyze", "evaluate", "compare", "assess", "consider", - "why does", "what if", "how would", "what are the", "which approach", - "think about", "design", "most likely", "reconstruct", "verify", - "assumption", "hypothesis", "compare and contrast", "weigh the options", - "recommend one", "given these constraints", "under these constraints", -} - // TechnicalTerms: architecture/distributed/security/infrastructure signals var technicalKeywords = []string{ "architecture", "distributed", "encryption", "authentication", "scalability", @@ -86,46 +74,9 @@ var simpleKeywords = []string{ "short", "quick", "beginner", "basic", "concise", } -// --- Output complexity keywords --- - -var enumTriggers = []string{ - "list every", "list all", "enumerate all", "all possible", - "every single", "show all", "name all", "give me all", -} - -var comprehensivenessMarkers = []string{ - "comprehensive", "exhaustive", "complete list", "full list", - "in detail", "detailed breakdown", "thorough", "in-depth", -} - -var elaborationMarkers = []string{ - "and what it does", "explain each", "describe each", "for each", - "with examples", "with descriptions", "along with", -} - -var limitingQualifiers = []string{ - "briefly", "top 3", "top 5", "top 10", "in one sentence", - "quickly", "summarize", "just the", "only the", "keep it short", - "tl;dr", "tldr", -} - -var referentialPhrases = []string{ +var continuationPhrases = []string{ "do it", "try again", "continue", "go ahead", "proceed", - "that one", "this one", "same thing", "again", "retry", + "that one", "this one", "same thing", "again", "yes do that", "go with that", "use option 1", "use option 2", "use option 3", "now write it", } - -var referentialReferenceWords = []string{ - "it", "this", "that", "same", "previous", "earlier", -} - -var referentialActionWords = []string{ - "do", "retry", "continue", "proceed", "use", "fix", - "rewrite", "shorten", "clean", "adjust", "make", "give", "answer", -} - -var taskShiftPhrases = []string{ - "translate", "summarize", "in one sentence", "one sentence", - "in spanish", "in french", "in german", "more politely", "more polite", -} diff --git a/plugins/governance/complexity/matcher.go b/plugins/governance/complexity/matcher.go index 325721e3e19..c4a1dca19e9 100644 --- a/plugins/governance/complexity/matcher.go +++ b/plugins/governance/complexity/matcher.go @@ -1,6 +1,10 @@ package complexity -import "strings" +import ( + "strings" + + "github.com/blevesearch/go-porterstemmer" +) type compiledKeywordMask uint16 @@ -10,23 +14,25 @@ const ( maskStrongReasoning maskTechnical maskSimple - maskEnum - maskComprehensive - maskElaboration - maskLimiter - maskReferentialPhrase - maskReferentialReference - maskReferentialAction - maskTaskShift + maskContinuation ) const ( - lastTextBaseScanMask = maskCode | maskReasoning | maskStrongReasoning | maskTechnical | maskSimple | maskEnum | maskComprehensive | maskElaboration | maskLimiter - lastTextFullScanMask = lastTextBaseScanMask | maskReferentialPhrase | maskReferentialReference | maskReferentialAction | maskTaskShift - systemTextScanMask = maskCode | maskTechnical | maskSimple + lastTextBaseScanMask = maskCode | maskReasoning | maskStrongReasoning | maskTechnical | maskSimple | maskContinuation + systemTextScanMask = maskCode | maskTechnical contextTextScanMask = maskCode | maskReasoning | maskTechnical ) +const ( + // matchedKeywordStackPrealloc keeps the common dedupe path allocation-free. + // It is not a match limit; the slice grows if a prompt matches more keywords. + matchedKeywordStackPrealloc = 16 + + // Stems are built from word tokens, so NUL cannot appear in a token. This + // keeps compiled phrase keys unambiguous without escaping. + stemmedKeywordKeySeparator = "\x00" +) + type keywordMatchMode uint8 const ( @@ -36,34 +42,43 @@ const ( ) type compiledKeyword struct { + id int text string mask compiledKeywordMask matchMode keywordMatchMode } +type compiledStemmedKeyword struct { + matches []compiledStemmedKeywordMatch + stems []string + mask compiledKeywordMask +} + +// compiledStemmedKeywordMatch keeps the original keyword identity behind a +// stem sequence. Multiple configured keywords can collapse to the same stems +// but still need separate dedupe and scoring masks. +type compiledStemmedKeywordMatch struct { + id int + mask compiledKeywordMask +} + // compiledKeywordMatcher groups keywords by match strategy so request-time // scans can skip repeated per-keyword boundary-mode decisions. type compiledKeywordMatcher struct { wholeWordKeywords []compiledKeyword boundarySubstringKeywords []compiledKeyword plainSubstringKeywords []compiledKeyword + stemmedKeywordIndex map[string][]compiledStemmedKeyword } type textSignalCounts struct { - wordCount int - codeCount int - reasoningCount int - strongReasoningCount int - technicalCount int - simpleCount int - enumCount int - comprehensiveCount int - elaborationCount int - limitingQualifierCount int - referentialPhraseCount int - referentialReferenceCount int - referentialActionCount int - taskShiftCount int + wordCount int + codeCount int + reasoningCount int + strongReasoningCount int + technicalCount int + simpleCount int + continuationPhraseCount int } func newCompiledKeywordMatcher(keywords KeywordConfig) *compiledKeywordMatcher { @@ -77,6 +92,7 @@ func newCompiledKeywordMatcher(keywords KeywordConfig) *compiledKeywordMatcher { entry, ok := entries[text] if !ok { entry = compiledKeyword{ + id: len(entries), text: text, mask: mask, matchMode: keywordMatchModeFor(text), @@ -90,19 +106,13 @@ func newCompiledKeywordMatcher(keywords KeywordConfig) *compiledKeywordMatcher { addKeywords(keywords.CodeKeywords, maskCode) addKeywords(keywords.StrongReasoningKeywords, maskReasoning|maskStrongReasoning) - addKeywords(keywords.WeakReasoningKeywords, maskReasoning) addKeywords(keywords.TechnicalKeywords, maskTechnical) addKeywords(keywords.SimpleKeywords, maskSimple) - addKeywords(keywords.EnumTriggers, maskEnum) - addKeywords(keywords.ComprehensivenessMarkers, maskComprehensive) - addKeywords(keywords.ElaborationMarkers, maskElaboration) - addKeywords(keywords.LimitingQualifiers, maskLimiter) - addKeywords(keywords.ReferentialPhrases, maskReferentialPhrase) - addKeywords(keywords.ReferentialReferenceWords, maskReferentialReference) - addKeywords(keywords.ReferentialActionWords, maskReferentialAction) - addKeywords(keywords.TaskShiftPhrases, maskTaskShift) + addKeywords(keywords.ContinuationPhrases, maskContinuation) matcher := &compiledKeywordMatcher{} + var stemmedKeywords []compiledStemmedKeyword + stemmedByKey := make(map[string]int) for _, entry := range entries { switch entry.matchMode { case matchModeWholeWord: @@ -112,6 +122,38 @@ func newCompiledKeywordMatcher(keywords KeywordConfig) *compiledKeywordMatcher { case matchModePlainSubstring: matcher.plainSubstringKeywords = append(matcher.plainSubstringKeywords, entry) } + + // Stem matching is additive to the exact matcher above. Only normal + // word-token keywords and phrases are indexed; punctuation-heavy terms + // such as "ci/cd" stay on the literal matching path. + if stems, ok := stemKeywordTokens(entry.text); ok { + key := strings.Join(stems, stemmedKeywordKeySeparator) + if idx, exists := stemmedByKey[key]; exists { + stemmedKeywords[idx].matches = append(stemmedKeywords[idx].matches, compiledStemmedKeywordMatch{ + id: entry.id, + mask: entry.mask, + }) + stemmedKeywords[idx].mask |= entry.mask + continue + } + stemmedByKey[key] = len(stemmedKeywords) + stemmedKeywords = append(stemmedKeywords, compiledStemmedKeyword{ + matches: []compiledStemmedKeywordMatch{{ + id: entry.id, + mask: entry.mask, + }}, + stems: stems, + mask: entry.mask, + }) + } + } + if len(stemmedKeywords) > 0 { + matcher.stemmedKeywordIndex = make(map[string][]compiledStemmedKeyword, len(stemmedKeywords)) + for _, keyword := range stemmedKeywords { + // Index by the first stem so request-time matching checks only + // candidates that can start at the current request token. + matcher.stemmedKeywordIndex[keyword.stems[0]] = append(matcher.stemmedKeywordIndex[keyword.stems[0]], keyword) + } } return matcher } @@ -139,15 +181,21 @@ func (m *compiledKeywordMatcher) analyzeText(text string, scanMask compiledKeywo signals := textSignalCounts{ wordCount: countWordsNoAlloc(text), } + var matchedKeywordIDs [matchedKeywordStackPrealloc]int + matchedIDs := matchedKeywordIDs[:0] + recordMatch := func(keyword compiledKeyword) { + signals.addMask(keyword.mask) + matchedIDs = append(matchedIDs, keyword.id) + } if len(lowerText) >= wordPresenceSetMinBytes { - wordPresence := buildWordPresenceSet(lowerText) + wordPresence := buildWordPresenceSet(lowerText, signals.wordCount) for _, keyword := range m.wholeWordKeywords { if keyword.mask&scanMask == 0 { continue } if _, ok := wordPresence[keyword.text]; ok { - signals.addMask(keyword.mask) + recordMatch(keyword) } } } else { @@ -156,7 +204,7 @@ func (m *compiledKeywordMatcher) analyzeText(text string, scanMask compiledKeywo continue } if containsWord(lowerText, keyword.text) { - signals.addMask(keyword.mask) + recordMatch(keyword) } } } @@ -165,7 +213,7 @@ func (m *compiledKeywordMatcher) analyzeText(text string, scanMask compiledKeywo continue } if containsWord(lowerText, keyword.text) { - signals.addMask(keyword.mask) + recordMatch(keyword) } } for _, keyword := range m.plainSubstringKeywords { @@ -173,13 +221,53 @@ func (m *compiledKeywordMatcher) analyzeText(text string, scanMask compiledKeywo continue } if strings.Contains(lowerText, keyword.text) { - signals.addMask(keyword.mask) + recordMatch(keyword) } } + m.addStemmedMatches(lowerText, scanMask, matchedIDs, &signals) return signals } +func (m *compiledKeywordMatcher) addStemmedMatches(lowerText string, scanMask compiledKeywordMask, exactMatchedIDs []int, signals *textSignalCounts) { + if len(m.stemmedKeywordIndex) == 0 { + return + } + + requestStems := stemTextTokens(lowerText, signals.wordCount) + if len(requestStems) == 0 { + return + } + + var stemMatchedKeywordIDs [matchedKeywordStackPrealloc]int + stemMatchedIDs := stemMatchedKeywordIDs[:0] + for idx, stem := range requestStems { + for _, keyword := range m.stemmedKeywordIndex[stem] { + if keyword.mask&scanMask == 0 { + continue + } + if stemSequenceMatchesAt(requestStems, idx, keyword.stems) { + unmatchedMask := compiledKeywordMask(0) + for _, match := range keyword.matches { + if match.mask&scanMask == 0 { + continue + } + if keywordIDMatched(match.id, exactMatchedIDs) || keywordIDMatched(match.id, stemMatchedIDs) { + continue + } + // Exact matches win first; the stem path only contributes + // configured keyword IDs that have not already matched. + unmatchedMask |= match.mask + stemMatchedIDs = append(stemMatchedIDs, match.id) + } + if unmatchedMask != 0 { + signals.addMask(unmatchedMask) + } + } + } + } +} + // addMask increments every scoring bucket a matched keyword contributes to. func (s *textSignalCounts) addMask(mask compiledKeywordMask) { if mask&maskCode != 0 { @@ -197,36 +285,110 @@ func (s *textSignalCounts) addMask(mask compiledKeywordMask) { if mask&maskSimple != 0 { s.simpleCount++ } - if mask&maskEnum != 0 { - s.enumCount++ + if mask&maskContinuation != 0 { + s.continuationPhraseCount++ } - if mask&maskComprehensive != 0 { - s.comprehensiveCount++ +} + +// buildWordPresenceSet tokenizes large inputs once so whole-word matches become +// set lookups instead of repeated boundary-aware scans. +func buildWordPresenceSet(text string, capacityHint int) map[string]struct{} { + words := make(map[string]struct{}, capacityHint) + start := -1 + for i, r := range text { + if isWordChar(r) { + if start == -1 { + start = i + } + continue + } + if start != -1 { + words[text[start:i]] = struct{}{} + start = -1 + } } - if mask&maskElaboration != 0 { - s.elaborationCount++ + if start != -1 { + words[text[start:]] = struct{}{} } - if mask&maskLimiter != 0 { - s.limitingQualifierCount++ + return words +} + +func stemKeywordTokens(keyword string) ([]string, bool) { + tokens, ok := tokenizeStemEligibleText(keyword) + if !ok { + return nil, false } - if mask&maskReferentialPhrase != 0 { - s.referentialPhraseCount++ + return stemTokens(tokens), true +} + +// stemTextTokens tokenizes request text with the same word-character rules used +// by exact whole-word matching, then stems those tokens for the additive pass. +func stemTextTokens(text string, capacityHint int) []string { + tokens := tokenizeWordText(text, capacityHint) + if len(tokens) == 0 { + return nil } - if mask&maskReferentialReference != 0 { - s.referentialReferenceCount++ + return stemTokens(tokens) +} + +// stemTokens receives already-lowercased tokens. StemWithoutLowerCasing avoids +// redoing lowercase work that analyzeText has already performed. +func stemTokens(tokens []string) []string { + stems := make([]string, 0, len(tokens)) + var runeBuffer []rune + for _, token := range tokens { + stem := token + if hasMoreThanTwoRunes(token) { + runeBuffer = runeBuffer[:0] + for _, r := range token { + runeBuffer = append(runeBuffer, r) + } + stemRunes := porterstemmer.StemWithoutLowerCasing(runeBuffer) + if !runesEqualString(stemRunes, token) { + stem = string(stemRunes) + } + } + if stem == "" { + continue + } + stems = append(stems, stem) } - if mask&maskReferentialAction != 0 { - s.referentialActionCount++ + return stems +} + +func hasMoreThanTwoRunes(text string) bool { + count := 0 + for range text { + count++ + if count > 2 { + return true + } } - if mask&maskTaskShift != 0 { - s.taskShiftCount++ + return false +} + +// tokenizeStemEligibleText accepts only space-separated word tokens. This keeps +// symbol/punctuation terms on the existing literal matcher instead of inventing +// surprising stemmed behavior for terms like "ci/cd". +func tokenizeStemEligibleText(text string) ([]string, bool) { + fields := strings.Fields(text) + if len(fields) == 0 { + return nil, false } + for _, field := range fields { + for _, r := range field { + if !isWordChar(r) { + return nil, false + } + } + } + return fields, true } -// buildWordPresenceSet tokenizes large inputs once so whole-word matches become -// set lookups instead of repeated boundary-aware scans. -func buildWordPresenceSet(text string) map[string]struct{} { - words := make(map[string]struct{}, 64) +// tokenizeWordText extracts the request tokens used for stem matching. The +// capacity hint is the word count already computed for scoring. +func tokenizeWordText(text string, capacityHint int) []string { + tokens := make([]string, 0, capacityHint) start := -1 for i, r := range text { if isWordChar(r) { @@ -236,12 +398,44 @@ func buildWordPresenceSet(text string) map[string]struct{} { continue } if start != -1 { - words[text[start:i]] = struct{}{} + tokens = append(tokens, text[start:i]) start = -1 } } if start != -1 { - words[text[start:]] = struct{}{} + tokens = append(tokens, text[start:]) } - return words + return tokens +} + +func stemSequenceMatchesAt(tokens []string, start int, sequence []string) bool { + if len(sequence) == 0 || start+len(sequence) > len(tokens) { + return false + } + for idx, stem := range sequence { + if tokens[start+idx] != stem { + return false + } + } + return true +} + +func runesEqualString(runes []rune, text string) bool { + idx := 0 + for _, r := range text { + if idx >= len(runes) || runes[idx] != r { + return false + } + idx++ + } + return idx == len(runes) +} + +func keywordIDMatched(id int, matchedIDs []int) bool { + for _, matchedID := range matchedIDs { + if id == matchedID { + return true + } + } + return false } diff --git a/plugins/governance/complexity/matcher_test.go b/plugins/governance/complexity/matcher_test.go new file mode 100644 index 00000000000..dd2a992b4eb --- /dev/null +++ b/plugins/governance/complexity/matcher_test.go @@ -0,0 +1,100 @@ +package complexity + +import "testing" + +func TestCompiledKeywordMatcher_StemmedSingleWordMatchesInflection(t *testing.T) { + matcher := newCompiledKeywordMatcher(KeywordConfig{ + CodeKeywords: []string{"debug"}, + }) + + signals := matcher.analyzeText("We are debugging the handler before release.", lastTextBaseScanMask) + if signals.codeCount != 1 { + t.Fatalf("expected stemmed debug keyword to match debugging once, got codeCount=%d", signals.codeCount) + } +} + +func TestCompiledKeywordMatcher_CustomInflectedKeywordMatchesRootForm(t *testing.T) { + matcher := newCompiledKeywordMatcher(KeywordConfig{ + CodeKeywords: []string{"debugging"}, + }) + + signals := matcher.analyzeText("Please debug the handler before release.", lastTextBaseScanMask) + if signals.codeCount != 1 { + t.Fatalf("expected custom debugging keyword to match debug once, got codeCount=%d", signals.codeCount) + } +} + +func TestCompiledKeywordMatcher_ExactAndStemmedMatchDoesNotDoubleCount(t *testing.T) { + matcher := newCompiledKeywordMatcher(KeywordConfig{ + CodeKeywords: []string{"debug"}, + }) + + signals := matcher.analyzeText("Please debug the handler before release.", lastTextBaseScanMask) + if signals.codeCount != 1 { + t.Fatalf("expected exact debug match not to be double-counted by stemming, got codeCount=%d", signals.codeCount) + } +} + +func TestCompiledKeywordMatcher_RepeatedStemmedFormsDoNotDoubleCount(t *testing.T) { + matcher := newCompiledKeywordMatcher(KeywordConfig{ + CodeKeywords: []string{"debug"}, + }) + + signals := matcher.analyzeText("Please debugged and debugging the handler before release.", lastTextBaseScanMask) + if signals.codeCount != 1 { + t.Fatalf("expected repeated stemmed forms of debug to count once, got codeCount=%d", signals.codeCount) + } +} + +func TestCompiledKeywordMatcher_StemDedupePreservesDistinctSameStemKeywords(t *testing.T) { + matcher := newCompiledKeywordMatcher(KeywordConfig{ + CodeKeywords: []string{"debug"}, + TechnicalKeywords: []string{"debugging"}, + }) + + signals := matcher.analyzeText("Please debug the handler before release.", lastTextBaseScanMask) + if signals.codeCount != 1 { + t.Fatalf("expected exact debug keyword to contribute code once, got codeCount=%d", signals.codeCount) + } + if signals.technicalCount != 1 { + t.Fatalf("expected distinct same-stem debugging keyword to contribute technical once, got technicalCount=%d", signals.technicalCount) + } +} + +func TestCompiledKeywordMatcher_StemmedPhraseMatchesContiguousVariant(t *testing.T) { + matcher := newCompiledKeywordMatcher(KeywordConfig{ + CodeKeywords: []string{"analyzing this function"}, + }) + + signals := matcher.analyzeText("Please analyze this function before merging.", lastTextBaseScanMask) + if signals.codeCount != 1 { + t.Fatalf("expected stemmed phrase to match contiguous variant once, got codeCount=%d", signals.codeCount) + } +} + +func TestCompiledKeywordMatcher_StemmedPhraseDoesNotMatchInsertedWords(t *testing.T) { + matcher := newCompiledKeywordMatcher(KeywordConfig{ + CodeKeywords: []string{"analyzing this function"}, + }) + + signals := matcher.analyzeText("Please analyze this Python function before merging.", lastTextBaseScanMask) + if signals.codeCount != 0 { + t.Fatalf("expected stemmed phrase not to match with inserted words, got codeCount=%d", signals.codeCount) + } +} + +func TestCompiledKeywordMatcher_PunctuationKeywordRemainsLiteral(t *testing.T) { + matcher := newCompiledKeywordMatcher(KeywordConfig{ + CodeKeywords: []string{"ci/cd"}, + }) + + literalSignals := matcher.analyzeText("The ci/cd pipeline is failing.", lastTextBaseScanMask) + if literalSignals.codeCount != 1 { + t.Fatalf("expected literal ci/cd keyword to match, got codeCount=%d", literalSignals.codeCount) + } + + tokenSignals := matcher.analyzeText("The ci cd pipeline is failing.", lastTextBaseScanMask) + if tokenSignals.codeCount != 0 { + t.Fatalf("expected ci cd not to match literal ci/cd keyword, got codeCount=%d", tokenSignals.codeCount) + } +} diff --git a/plugins/governance/complexity/utils.go b/plugins/governance/complexity/utils.go index b6fbd6c07f6..06047b86b11 100644 --- a/plugins/governance/complexity/utils.go +++ b/plugins/governance/complexity/utils.go @@ -72,24 +72,6 @@ func scoreCount(count, capAt int) float64 { return math.Min(1.0, float64(count)/float64(capAt)) } -func scoreOutputComplexity(signals textSignalCounts) float64 { - totalCount := signals.enumCount + signals.comprehensiveCount + signals.elaborationCount - if totalCount == 0 { - return 0.0 - } - - enumScore := math.Min(1.0, float64(signals.enumCount)) - compScore := math.Min(1.0, float64(signals.comprehensiveCount)) - elabScore := math.Min(1.0, float64(signals.elaborationCount)) - - rawScore := (enumScore * 0.4) + (compScore * 0.3) + (elabScore * 0.3) - if signals.limitingQualifierCount > 0 { - rawScore *= 0.3 - } - - return math.Min(1.0, rawScore) -} - // scoreTokenCount scores based on word count of the text. func scoreTokenCount(words int) float64 { switch { diff --git a/plugins/governance/go.mod b/plugins/governance/go.mod index eea6b691abb..52076371865 100644 --- a/plugins/governance/go.mod +++ b/plugins/governance/go.mod @@ -5,6 +5,7 @@ go 1.26.4 require gorm.io/gorm v1.31.1 require ( + github.com/blevesearch/go-porterstemmer v1.0.3 github.com/google/cel-go v0.28.1 github.com/google/uuid v1.6.0 github.com/maximhq/bifrost/core v1.6.2 diff --git a/plugins/governance/go.sum b/plugins/governance/go.sum index 1665c7de2f0..7b08ab04eee 100644 --- a/plugins/governance/go.sum +++ b/plugins/governance/go.sum @@ -87,6 +87,8 @@ github.com/aws/smithy-go v1.27.1 h1:4T340VFndXtADGF52gYa1POyL7s9E4Z1OeZ1hCscIw8= github.com/aws/smithy-go v1.27.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc= github.com/bahlo/generic-list-go v0.2.0 h1:5sz/EEAK+ls5wF+NeqDpk5+iNdMDXrh3z3nPnH1Wvgk= github.com/bahlo/generic-list-go v0.2.0/go.mod h1:2KvAjgMlE5NNynlg/5iLrrCCZ2+5xWbdbCW3pNTGyYg= +github.com/blevesearch/go-porterstemmer v1.0.3 h1:GtmsqID0aZdCSNiY8SkuPJ12pD4jI+DdXTAn4YRcHCo= +github.com/blevesearch/go-porterstemmer v1.0.3/go.mod h1:angGc5Ht+k2xhJdZi511LtmxuEf0OVpvUUNrwmM1P7M= github.com/bmatcuk/doublestar v1.1.1/go.mod h1:UD6OnuiIn0yFxxA2le/rnRU1G4RaI4UvFv1sNto9p6w= github.com/bsm/ginkgo/v2 v2.12.0 h1:Ny8MWAHyOepLGlLKYmXG4IEkioBysk6GpaRTLC8zwWs= github.com/bsm/ginkgo/v2 v2.12.0/go.mod h1:SwYbGRRDovPVboqFv0tPTcG1sN61LM1Z4ARdbAV9g4c= diff --git a/plugins/governance/main.go b/plugins/governance/main.go index a91dd75337c..b32538a749d 100644 --- a/plugins/governance/main.go +++ b/plugins/governance/main.go @@ -29,6 +29,8 @@ const ( governanceRejectedContextKey schemas.BifrostContextKey = "bf-governance-rejected" VirtualKeyPrefix = "sk-bf-" + + noComplexitySignalLog = "Complexity analysis skipped: no configured complexity signal matched the latest user message; continuing with existing routing path" ) // Config is the configuration for the governance plugin @@ -700,6 +702,13 @@ func (p *GovernancePlugin) applyRoutingRules(ctx *schemas.BifrostContext, req *s } result := analyzer.Analyze(input) + if result == nil { + if p.logger != nil { + p.logger.Debug("[Governance] %s", noComplexitySignalLog) + } + ctx.AppendRoutingEngineLog(schemas.RoutingEngineRoutingRule, schemas.LogLevelDebug, noComplexitySignalLog) + return nil + } if p.logger != nil { p.logger.Debug( "[Governance] Complexity analysis details: tier=%s score=%.2f words=%d", @@ -1195,7 +1204,7 @@ func (p *GovernancePlugin) PreRequestHook(ctx *schemas.BifrostContext, req *sche if virtualKeyValue != "" { var ok bool virtualKey, ok = p.store.GetVirtualKey(ctx, virtualKeyValue) - if !ok || virtualKey == nil || !virtualKey.IsActiveValue() { + if !ok || virtualKey == nil || !virtualKey.IsActiveValue() || virtualKey.IsExpiredAt(time.Now().UTC()) { return nil } } @@ -1431,7 +1440,7 @@ func (p *GovernancePlugin) PreMCPHook(ctx *schemas.BifrostContext, req *schemas. // This runs independently of EvaluateGovernanceRequest to enforce execution-time allow-list. if virtualKeyValue != "" { vk, ok := p.store.GetVirtualKey(ctx, virtualKeyValue) - if !ok || vk == nil || !vk.IsActiveValue() { + if !ok || vk == nil { // VK became invalid after initial check - fail closed for security ctx.SetValue(governanceRejectedContextKey, true) return req, &schemas.MCPPluginShortCircuit{Error: &schemas.BifrostError{ @@ -1442,6 +1451,26 @@ func (p *GovernancePlugin) PreMCPHook(ctx *schemas.BifrostContext, req *schemas. }, }}, nil } + if !vk.IsActiveValue() { + ctx.SetValue(governanceRejectedContextKey, true) + return req, &schemas.MCPPluginShortCircuit{Error: &schemas.BifrostError{ + Type: bifrost.Ptr(string(DecisionVirtualKeyBlocked)), + StatusCode: bifrost.Ptr(403), + Error: &schemas.ErrorField{ + Message: "Virtual key is inactive", + }, + }}, nil + } + if vk.IsExpiredAt(time.Now().UTC()) { + ctx.SetValue(governanceRejectedContextKey, true) + return req, &schemas.MCPPluginShortCircuit{Error: &schemas.BifrostError{ + Type: bifrost.Ptr(string(DecisionVirtualKeyBlocked)), + StatusCode: bifrost.Ptr(403), + Error: &schemas.ErrorField{ + Message: "Virtual key has expired", + }, + }}, nil + } if !p.isMCPToolAllowedByVK(vk, toolName) { ctx.SetValue(governanceRejectedContextKey, true) return req, &schemas.MCPPluginShortCircuit{Error: &schemas.BifrostError{ diff --git a/plugins/governance/ratelimitreset_test.go b/plugins/governance/ratelimitreset_test.go new file mode 100644 index 00000000000..b83977983a0 --- /dev/null +++ b/plugins/governance/ratelimitreset_test.go @@ -0,0 +1,203 @@ +package governance + +import ( + "context" + "fmt" + "testing" + "time" + + configstoreTables "github.com/maximhq/bifrost/framework/configstore/tables" +) + +const resetBenchmarkVirtualKeys = 5000 + +// TestRequestTimeRateLimitResetPerformance verifies request-time rate-limit +// reset stays constant-time with thousands of embedded references. Runs as a +// regular test so it executes in CI via go test / make test-governance. +func TestRequestTimeRateLimitResetPerformance(t *testing.T) { + ctx := context.Background() + store := newStandaloneStoreForResetBenchmark() + rateLimitID := seedResetBenchmarkVirtualKeys(ctx, store, resetBenchmarkVirtualKeys) + + const iterations = 1000 + start := time.Now() + for i := 0; i < iterations; i++ { + markRequestRateLimitExpired(store, rateLimitID) + if err := store.BumpRateLimitUsage(ctx, rateLimitID, 0, false, true); err != nil { + t.Fatal(err) + } + } + nsPerOp := float64(time.Since(start).Nanoseconds()) / float64(iterations) + // Request-time reset must stay O(1). With the reference-refresh regression + // this exceeds 500µs per op; the fix keeps it well under 100µs. + const maxNsPerOp = 100_000 // 100µs + if nsPerOp > maxNsPerOp { + t.Fatalf("request-time reset too slow: %.0f ns/op (max %d ns/op) — likely running expensive O(N) work on the request hot path", nsPerOp, maxNsPerOp) + } + t.Logf("request-time reset: %.0f ns/op (%d iterations)", nsPerOp, iterations) +} + +// BenchmarkSingleRequestTimeRateLimitResetDoesNotRefreshReferences runs the same +// reset path exactly once per benchmark iteration for fixed-count profiles. +func BenchmarkSingleRequestTimeRateLimitResetDoesNotRefreshReferences(b *testing.B) { + ctx := context.Background() + + b.ReportAllocs() + for i := 0; i < b.N; i++ { + store := newStandaloneStoreForResetBenchmark() + rateLimitID := seedResetBenchmarkVirtualKeys(ctx, store, resetBenchmarkVirtualKeys) + markRequestRateLimitExpired(store, rateLimitID) + + b.StartTimer() + if err := store.BumpRateLimitUsage(ctx, rateLimitID, 0, false, true); err != nil { + b.Fatal(err) + } + b.StopTimer() + } +} + +// TestRequestTimeRateLimitResetSkipsReferenceRefresh verifies request-time resets +// keep canonical side effects without scanning embedded references. +func TestRequestTimeRateLimitResetSkipsReferenceRefresh(t *testing.T) { + ctx := context.Background() + store := newStandaloneStoreForResetBenchmark() + rateLimitID := seedResetBenchmarkVirtualKeys(ctx, store, 10) + staleRateLimit := store.LoadRateLimit(ctx, rateLimitID) + markRequestRateLimitExpired(store, rateLimitID) + + resetHookCalls := 0 + store.SetResetHooks(nil, func(resetRateLimits []*configstoreTables.TableRateLimit) { + resetHookCalls++ + if len(resetRateLimits) != 1 { + t.Fatalf("expected one reset rate limit, got %d", len(resetRateLimits)) + } + if resetRateLimits[0].ID != rateLimitID { + t.Fatalf("expected reset rate limit %q, got %q", rateLimitID, resetRateLimits[0].ID) + } + }) + + if err := store.BumpRateLimitUsage(ctx, rateLimitID, 0, false, true); err != nil { + t.Fatal(err) + } + resetRateLimit := store.LoadRateLimit(ctx, rateLimitID) + if resetRateLimit == nil { + t.Fatal("expected reset rate limit to remain loaded") + } + if resetRateLimit.RequestCurrentUsage != 1 { + t.Fatalf("expected request usage to be bumped after reset, got %d", resetRateLimit.RequestCurrentUsage) + } + if resetRateLimit.RequestLastReset.Before(staleRateLimit.RequestLastReset) || resetRateLimit.RequestLastReset.Equal(staleRateLimit.RequestLastReset) { + t.Fatalf("expected request reset timestamp to advance beyond %s, got %s", staleRateLimit.RequestLastReset, resetRateLimit.RequestLastReset) + } + if resetHookCalls != 1 { + t.Fatalf("expected request-time reset hook to fire once, got %d", resetHookCalls) + } + + store.LastDBUsagesRateLimitsRequestsMu.RLock() + lastDBRequests := store.LastDBUsagesRequestsRateLimits[rateLimitID] + store.LastDBUsagesRateLimitsRequestsMu.RUnlock() + if lastDBRequests != 0 { + t.Fatalf("expected request LastDB baseline to reset to 0, got %d", lastDBRequests) + } + + rawVK, ok := store.virtualKeys.Load("sk-bf-reset-00000") + if !ok || rawVK == nil { + t.Fatal("expected seeded virtual key to remain loaded") + } + vk, ok := rawVK.(*configstoreTables.TableVirtualKey) + if !ok || vk == nil { + t.Fatal("expected seeded virtual key to have the correct type") + } + // Embedded reference should remain stale (not refreshed) after request-time reset. + // It may be non-nil because CreateVirtualKeyInMemory keeps embedded references, + // but it must NOT point at the freshly reset canonical object. + if vk.RateLimit == resetRateLimit { + t.Fatal("expected request-time reset NOT to refresh embedded virtual-key rate-limit reference") + } + if vk.RateLimitID == nil || *vk.RateLimitID != rateLimitID { + t.Fatal("expected request-time reset to preserve owner-to-rate-limit ID mapping") + } +} + +// TestBackgroundRateLimitResetRefreshesReferences verifies non-request resets still hydrate embedded references. +func TestBackgroundRateLimitResetRefreshesReferences(t *testing.T) { + ctx := context.Background() + store := newStandaloneStoreForResetBenchmark() + rateLimitID := seedResetBenchmarkVirtualKeys(ctx, store, 10) + staleRateLimit := store.LoadRateLimit(ctx, rateLimitID) + markRequestRateLimitExpired(store, rateLimitID) + + resetRateLimits := store.ResetExpiredRateLimitsInMemory(ctx, true, rateLimitID) + if len(resetRateLimits) != 1 { + t.Fatalf("expected one background reset, got %d", len(resetRateLimits)) + } + + rawVK, ok := store.virtualKeys.Load("sk-bf-reset-00000") + if !ok || rawVK == nil { + t.Fatal("expected seeded virtual key to remain loaded") + } + vk, ok := rawVK.(*configstoreTables.TableVirtualKey) + if !ok || vk == nil { + t.Fatal("expected seeded virtual key to have the correct type") + } + if vk.RateLimit == nil { + t.Fatal("expected background reset to refresh virtual-key rate-limit reference") + } + if vk.RateLimit != resetRateLimits[0] { + t.Fatal("expected background reset to point virtual-key reference at canonical reset object") + } + if vk.RateLimitID == nil || *vk.RateLimitID != rateLimitID { + t.Fatal("expected background reset to preserve owner-to-rate-limit ID mapping") + } + if resetRateLimits[0] == staleRateLimit { + t.Fatal("expected canonical reset to replace stale rate-limit snapshot") + } +} + +// newStandaloneStoreForResetBenchmark creates an in-memory-only store so benchmark +// profiles isolate sync.Map/reference-update CPU rather than DB IO. +func newStandaloneStoreForResetBenchmark() *LocalGovernanceStore { + return &LocalGovernanceStore{ + logger: NewMockLogger(), + LastDBUsagesBudgets: map[string]float64{}, + LastDBUsagesTokensRateLimits: map[string]int64{}, + LastDBUsagesRequestsRateLimits: map[string]int64{}, + } +} + +// seedResetBenchmarkVirtualKeys creates many VK owner mappings to one rate limit so +// the benchmark guards against reintroducing global owner scans. +func seedResetBenchmarkVirtualKeys(ctx context.Context, store *LocalGovernanceStore, virtualKeys int) string { + rateLimitID := "request-reset-rate-limit" + rateLimit := buildRateLimit(rateLimitID, 1_000_000_000, 1_000_000_000) + store.rateLimits.Store(rateLimitID, rateLimit) + + for i := 0; i < virtualKeys; i++ { + vk := buildVirtualKeyWithRateLimit( + fmt.Sprintf("reset-vk-%05d", i), + fmt.Sprintf("sk-bf-reset-%05d", i), + fmt.Sprintf("Reset VK %05d", i), + rateLimit, + ) + store.CreateVirtualKeyInMemory(ctx, vk) + } + + return rateLimitID +} + +// markRequestRateLimitExpired forces BumpRateLimitUsage down the request-time +// reset path on the next call. +func markRequestRateLimitExpired(store *LocalGovernanceStore, rateLimitID string) { + raw, ok := store.rateLimits.Load(rateLimitID) + if !ok || raw == nil { + return + } + rateLimit, ok := raw.(*configstoreTables.TableRateLimit) + if !ok || rateLimit == nil { + return + } + clone := *rateLimit + clone.RequestCurrentUsage = 123 + clone.RequestLastReset = time.Now().Add(-2 * time.Minute) + store.rateLimits.Store(rateLimitID, &clone) +} diff --git a/plugins/governance/resolver.go b/plugins/governance/resolver.go index 59cbc0aede7..d05f3971e7b 100644 --- a/plugins/governance/resolver.go +++ b/plugins/governance/resolver.go @@ -4,6 +4,7 @@ package governance import ( "context" "fmt" + "time" "github.com/maximhq/bifrost/core/schemas" configstoreTables "github.com/maximhq/bifrost/framework/configstore/tables" @@ -271,6 +272,12 @@ func (r *BudgetResolver) EvaluateVirtualKeyRequest(ctx *schemas.BifrostContext, Reason: "Virtual key is inactive", } } + if vk.IsExpiredAt(time.Now().UTC()) { + return &EvaluationResult{ + Decision: DecisionVirtualKeyBlocked, + Reason: "Virtual key has expired", + } + } // 2. Check provider filtering if requestType != schemas.MCPToolExecutionRequest && requestType != schemas.ListModelsRequest && !r.isProviderAllowed(vk, provider) { return &EvaluationResult{ @@ -408,16 +415,18 @@ func (r *BudgetResolver) isProviderAllowed(vk *configstoreTables.TableVirtualKey // checkRateLimitHierarchy checks provider-level rate limits first, then VK rate limits using flexible approach func (r *BudgetResolver) checkRateLimitHierarchy(ctx context.Context, vk *configstoreTables.TableVirtualKey, request *EvaluationRequest) *EvaluationResult { if decision, err := r.store.CheckVirtualKeyRateLimit(ctx, vk, request, nil, nil); err != nil || isRateLimitViolation(decision) { - // Check provider-level first (matching check order), then VK-level + // Check provider-level first (matching check order), then VK-level. + // Resolve by ID from the canonical rate-limit map because embedded + // references can intentionally remain stale after request-time resets. var rateLimitInfo *configstoreTables.TableRateLimit for _, pc := range vk.ProviderConfigs { - if pc.Provider == string(request.Provider) && pc.RateLimit != nil { - rateLimitInfo = pc.RateLimit + if pc.Provider == string(request.Provider) && pc.RateLimitID != nil { + rateLimitInfo = r.store.LoadRateLimit(ctx, *pc.RateLimitID) break } } - if rateLimitInfo == nil && vk.RateLimit != nil { - rateLimitInfo = vk.RateLimit + if rateLimitInfo == nil && vk.RateLimitID != nil { + rateLimitInfo = r.store.LoadRateLimit(ctx, *vk.RateLimitID) } return &EvaluationResult{ Decision: decision, @@ -478,7 +487,7 @@ func (r *BudgetResolver) isProviderRateLimitViolated(ctx context.Context, vk *co } // 2. Check VK-level provider config rate limit - if config.RateLimit == nil { + if config.RateLimitID == nil { return false } decision, err := r.store.CheckVirtualKeyRateLimit(ctx, vk, request, nil, nil) diff --git a/plugins/governance/resolver_test.go b/plugins/governance/resolver_test.go index 9e5fd94cad9..95de31b8608 100644 --- a/plugins/governance/resolver_test.go +++ b/plugins/governance/resolver_test.go @@ -221,7 +221,7 @@ func TestBudgetResolver_EvaluateRequest_RateLimitExpired(t *testing.T) { require.NoError(t, err) // Reset expired rate limits (simulating ticker behavior) - expiredRateLimits := store.ResetExpiredRateLimitsInMemory(context.Background()) + expiredRateLimits := store.ResetExpiredRateLimitsInMemory(context.Background(), true) err = store.ResetExpiredRateLimits(context.Background(), expiredRateLimits) require.NoError(t, err) @@ -576,3 +576,93 @@ func TestBudgetResolver_EvaluateRequest_PassthroughModelFiltering(t *testing.T) }) } } + +// TestBudgetResolver_EvaluateVirtualKeyRequest_ActiveNoExpiry verifies that a VK +// with no expiry is allowed. +func TestBudgetResolver_EvaluateVirtualKeyRequest_ActiveNoExpiry(t *testing.T) { + logger := NewMockLogger() + vk := buildVirtualKey("vk1", "sk-bf-test", "Test VK", true) + vk.ProviderConfigs = []configstoreTables.TableVirtualKeyProviderConfig{ + buildProviderConfig("openai", []string{"*"}), + } + + store, err := NewLocalGovernanceStore(context.Background(), logger, nil, &configstore.GovernanceConfig{ + VirtualKeys: []configstoreTables.TableVirtualKey{*vk}, + }, nil) + require.NoError(t, err) + + resolver := NewBudgetResolver(store, nil, logger, nil) + ctx := &schemas.BifrostContext{} + + result := resolver.EvaluateVirtualKeyRequest(ctx, "sk-bf-test", schemas.OpenAI, "gpt-4", schemas.ChatCompletionRequest, false) + assertDecision(t, DecisionAllow, result) +} + +// TestBudgetResolver_EvaluateVirtualKeyRequest_FutureExpiry verifies that a VK +// with a future expiry is allowed. +func TestBudgetResolver_EvaluateVirtualKeyRequest_FutureExpiry(t *testing.T) { + logger := NewMockLogger() + future := time.Now().UTC().Add(time.Hour) + vk := buildVirtualKey("vk1", "sk-bf-test", "Test VK", true) + vk.ExpiresAt = &future + vk.ProviderConfigs = []configstoreTables.TableVirtualKeyProviderConfig{ + buildProviderConfig("openai", []string{"*"}), + } + + store, err := NewLocalGovernanceStore(context.Background(), logger, nil, &configstore.GovernanceConfig{ + VirtualKeys: []configstoreTables.TableVirtualKey{*vk}, + }, nil) + require.NoError(t, err) + + resolver := NewBudgetResolver(store, nil, logger, nil) + ctx := &schemas.BifrostContext{} + + result := resolver.EvaluateVirtualKeyRequest(ctx, "sk-bf-test", schemas.OpenAI, "gpt-4", schemas.ChatCompletionRequest, false) + assertDecision(t, DecisionAllow, result) +} + +// TestBudgetResolver_EvaluateVirtualKeyRequest_ExpiredKey verifies that an +// active VK with a past expiry is blocked with DecisionVirtualKeyBlocked. +func TestBudgetResolver_EvaluateVirtualKeyRequest_ExpiredKey(t *testing.T) { + logger := NewMockLogger() + past := time.Now().UTC().Add(-time.Second) + vk := buildVirtualKey("vk1", "sk-bf-test", "Test VK", true) + vk.ExpiresAt = &past + vk.ProviderConfigs = []configstoreTables.TableVirtualKeyProviderConfig{ + buildProviderConfig("openai", []string{"*"}), + } + + store, err := NewLocalGovernanceStore(context.Background(), logger, nil, &configstore.GovernanceConfig{ + VirtualKeys: []configstoreTables.TableVirtualKey{*vk}, + }, nil) + require.NoError(t, err) + + resolver := NewBudgetResolver(store, nil, logger, nil) + ctx := &schemas.BifrostContext{} + + result := resolver.EvaluateVirtualKeyRequest(ctx, "sk-bf-test", schemas.OpenAI, "gpt-4", schemas.ChatCompletionRequest, false) + assertDecision(t, DecisionVirtualKeyBlocked, result) + assert.Contains(t, result.Reason, "expired") +} + +// TestBudgetResolver_EvaluateVirtualKeyRequest_InactiveWithFutureExpiry verifies +// that an inactive VK is blocked as inactive, not expired, even when it has a +// future expiry. +func TestBudgetResolver_EvaluateVirtualKeyRequest_InactiveWithFutureExpiry(t *testing.T) { + logger := NewMockLogger() + future := time.Now().UTC().Add(time.Hour) + vk := buildVirtualKey("vk1", "sk-bf-test", "Test VK", false) + vk.ExpiresAt = &future + + store, err := NewLocalGovernanceStore(context.Background(), logger, nil, &configstore.GovernanceConfig{ + VirtualKeys: []configstoreTables.TableVirtualKey{*vk}, + }, nil) + require.NoError(t, err) + + resolver := NewBudgetResolver(store, nil, logger, nil) + ctx := &schemas.BifrostContext{} + + result := resolver.EvaluateVirtualKeyRequest(ctx, "sk-bf-test", schemas.OpenAI, "gpt-4", schemas.ChatCompletionRequest, false) + assertDecision(t, DecisionVirtualKeyBlocked, result) + assert.Contains(t, result.Reason, "inactive") +} diff --git a/plugins/governance/store.go b/plugins/governance/store.go index 490c4dc0f8b..7bb405799d5 100644 --- a/plugins/governance/store.go +++ b/plugins/governance/store.go @@ -24,14 +24,19 @@ type EntityWiseRateLimits map[string][]*configstoreTables.TableRateLimit // LocalGovernanceStore provides in-memory cache for governance data with fast, non-blocking access type LocalGovernanceStore struct { // Core data maps using sync.Map for lock-free reads - virtualKeys sync.Map // string -> *VirtualKey (VK value -> VirtualKey with preloaded relationships) - teams sync.Map // string -> *Team (Team ID -> Team) - customers sync.Map // string -> *Customer (Customer ID -> Customer) - budgets sync.Map // string -> *Budget (Budget ID -> Budget) - rateLimits sync.Map // string -> *RateLimit (RateLimit ID -> RateLimit) - modelConfigs sync.Map // string -> *ModelConfig (key: "modelName" or "modelName:provider" -> ModelConfig) - providers sync.Map // string -> *Provider (Provider name -> Provider with preloaded relationships) - routingRules sync.Map // string -> []*TableRoutingRule (key: "scope:scopeID" -> rules, scopeID="" for global) + virtualKeys sync.Map // string -> *VirtualKey (VK value -> VirtualKey with preloaded relationships) + // virtualKeysByID is a secondary index over virtualKeys keyed by VK row ID, + // giving O(1) by-ID lookups (e.g. the /mcp JWT auth path) without an O(n) + // scan or a database read. Maintained in lock-step with virtualKeys via + // storeVirtualKey / deleteVirtualKeyByValue — never write it directly. + virtualKeysByID sync.Map // string -> *VirtualKey (VK row ID -> VirtualKey) + teams sync.Map // string -> *Team (Team ID -> Team) + customers sync.Map // string -> *Customer (Customer ID -> Customer) + budgets sync.Map // string -> *Budget (Budget ID -> Budget) + rateLimits sync.Map // string -> *RateLimit (RateLimit ID -> RateLimit) + modelConfigs sync.Map // string -> *ModelConfig (key: "modelName" or "modelName:provider" -> ModelConfig) + providers sync.Map // string -> *Provider (Provider name -> Provider with preloaded relationships) + routingRules sync.Map // string -> []*TableRoutingRule (key: "scope:scopeID" -> rules, scopeID="" for global) // Last DB usages for budgets and rate limits LastDBUsagesBudgetsMu sync.RWMutex // Last DB usages for budgets @@ -140,8 +145,8 @@ type GovernanceStore interface { UpdateScopedModelBudgetUsageInMemory(ctx context.Context, scope, scopeID, model string, provider schemas.ModelProvider, cost float64) error UpdateScopedModelRateLimitUsageInMemory(ctx context.Context, scope, scopeID, model string, provider schemas.ModelProvider, tokensUsed int64, shouldUpdateTokens bool, shouldUpdateRequests bool) error // In-memory reset checks (return items that need DB sync) - ResetExpiredRateLimitsInMemory(ctx context.Context, rateLimitIDs ...string) []*configstoreTables.TableRateLimit - ResetExpiredBudgetsInMemory(ctx context.Context, budgetIDs ...string) []*configstoreTables.TableBudget + ResetExpiredRateLimitsInMemory(ctx context.Context, refreshReferences bool, rateLimitIDs ...string) []*configstoreTables.TableRateLimit + ResetExpiredBudgetsInMemory(ctx context.Context, refreshReferences bool, budgetIDs ...string) []*configstoreTables.TableBudget // DB sync for expired items ResetExpiredRateLimits(ctx context.Context, resetRateLimits []*configstoreTables.TableRateLimit) error ResetExpiredBudgets(ctx context.Context, resetBudgets []*configstoreTables.TableBudget) error @@ -418,7 +423,7 @@ func (gs *LocalGovernanceStore) BumpBudgetUsage(ctx context.Context, budgetID st return nil } if gs.budgetResetTarget(old, time.Now()) != nil { - gs.ResetExpiredBudgetsInMemory(ctx, budgetID) + gs.ResetExpiredBudgetsInMemory(ctx, false, budgetID) continue } clone := *old @@ -446,7 +451,7 @@ func (gs *LocalGovernanceStore) BumpRateLimitUsage(ctx context.Context, rateLimi return nil } if tokenNewLastReset, requestNewLastReset := gs.rateLimitResetTargets(old, time.Now()); tokenNewLastReset != nil || requestNewLastReset != nil { - gs.ResetExpiredRateLimitsInMemory(ctx, rateLimitID) + gs.ResetExpiredRateLimitsInMemory(ctx, false, rateLimitID) continue } clone := *old @@ -874,6 +879,49 @@ func (gs *LocalGovernanceStore) GetVirtualKey(ctx context.Context, vkValue strin return vk, true } +// GetVirtualKeyByID retrieves a virtual key by its row ID (lock-free) with all +// relationships preloaded, via the ID-keyed secondary index. Mirrors +// GetVirtualKey (which is keyed by value); used by by-ID hot paths such as /mcp +// JWT auth to avoid a per-request database read. +func (gs *LocalGovernanceStore) GetVirtualKeyByID(ctx context.Context, vkID string) (*configstoreTables.TableVirtualKey, bool) { + value, exists := gs.virtualKeysByID.Load(vkID) + if !exists || value == nil { + return nil, false + } + vk, ok := value.(*configstoreTables.TableVirtualKey) + if !ok || vk == nil { + return nil, false + } + return vk, true +} + +// storeVirtualKey writes vk into both the value-keyed primary map and the +// ID-keyed secondary index, keeping the two in lock-step. Every writer to +// virtualKeys must go through here so the ID index never diverges. +func (gs *LocalGovernanceStore) storeVirtualKey(value string, vk *configstoreTables.TableVirtualKey) { + if value == "" { + if vk != nil { + gs.logger.Warn("skipping virtual key %s with unresolvable value (env/vault ref could not be resolved)", vk.ID) + } + return + } + gs.virtualKeys.Store(value, vk) + if vk != nil && vk.ID != "" { + gs.virtualKeysByID.Store(vk.ID, vk) + } +} + +// deleteVirtualKeyByValue removes the VK stored under value from both the +// primary map and the ID-keyed secondary index. +func (gs *LocalGovernanceStore) deleteVirtualKeyByValue(value string) { + if existing, ok := gs.virtualKeys.Load(value); ok { + if vk, ok := existing.(*configstoreTables.TableVirtualKey); ok && vk != nil && vk.ID != "" { + gs.virtualKeysByID.Delete(vk.ID) + } + } + gs.virtualKeys.Delete(value) +} + // CheckRateLimit checks rate limits for tokens and requests across categories func (gs *LocalGovernanceStore) CheckRateLimit(ctx context.Context, entityWiseRateLimits EntityWiseRateLimits, tokensBaselines map[string]int64, requestsBaselines map[string]int64) (Decision, error) { for entity, rateLimits := range entityWiseRateLimits { @@ -1756,7 +1804,7 @@ func (gs *LocalGovernanceStore) budgetResetTarget(budget *configstoreTables.Tabl } // resetExpiredBudgetFromSnapshot applies the local side effects for an expired budget snapshot. -func (gs *LocalGovernanceStore) resetExpiredBudgetFromSnapshot(ctx context.Context, budget *configstoreTables.TableBudget, now time.Time) *configstoreTables.TableBudget { +func (gs *LocalGovernanceStore) resetExpiredBudgetFromSnapshot(ctx context.Context, budget *configstoreTables.TableBudget, now time.Time, refreshReferences bool) *configstoreTables.TableBudget { newLastReset := gs.budgetResetTarget(budget, now) if newLastReset == nil { return nil @@ -1769,14 +1817,19 @@ func (gs *LocalGovernanceStore) resetExpiredBudgetFromSnapshot(ctx context.Conte gs.LastDBUsagesBudgetsMu.Lock() gs.LastDBUsagesBudgets[resetBudget.ID] = 0 gs.LastDBUsagesBudgetsMu.Unlock() - gs.updateBudgetReferences(ctx, resetBudget) + if refreshReferences { + gs.updateBudgetReferences(ctx, resetBudget) + } gs.logger.Debug(fmt.Sprintf("Reset budget %s (was %.2f, reset to 0)", resetBudget.ID, oldUsage)) return resetBudget } // ResetExpiredBudgetsInMemory checks and resets budgets that have exceeded their reset duration. // With no budgetIDs it scans every budget; with IDs it only checks those budgets. -func (gs *LocalGovernanceStore) ResetExpiredBudgetsInMemory(ctx context.Context, budgetIDs ...string) []*configstoreTables.TableBudget { +// refreshReferences controls whether embedded owner references (VK, team, customer) are updated +// after reset. Background/ticker callers pass true; request-time callers pass false to avoid +// an O(N) scan of all owners on the hot path. +func (gs *LocalGovernanceStore) ResetExpiredBudgetsInMemory(ctx context.Context, refreshReferences bool, budgetIDs ...string) []*configstoreTables.TableBudget { now := time.Now() var resetBudgets []*configstoreTables.TableBudget resetOne := func(value any) { @@ -1784,7 +1837,7 @@ func (gs *LocalGovernanceStore) ResetExpiredBudgetsInMemory(ctx context.Context, if !ok || budget == nil { return } - if resetBudget := gs.resetExpiredBudgetFromSnapshot(ctx, budget, now); resetBudget != nil { + if resetBudget := gs.resetExpiredBudgetFromSnapshot(ctx, budget, now, refreshReferences); resetBudget != nil { resetBudgets = append(resetBudgets, resetBudget) } } @@ -1841,7 +1894,7 @@ func (gs *LocalGovernanceStore) rateLimitResetTargets(rateLimit *configstoreTabl } // resetExpiredRateLimitFromSnapshot applies the local side effects for an expired rate-limit snapshot. -func (gs *LocalGovernanceStore) resetExpiredRateLimitFromSnapshot(ctx context.Context, rateLimit *configstoreTables.TableRateLimit, now time.Time) *configstoreTables.TableRateLimit { +func (gs *LocalGovernanceStore) resetExpiredRateLimitFromSnapshot(ctx context.Context, rateLimit *configstoreTables.TableRateLimit, now time.Time, refreshReferences bool) *configstoreTables.TableRateLimit { tokenNewLastReset, requestNewLastReset := gs.rateLimitResetTargets(rateLimit, now) if tokenNewLastReset == nil && requestNewLastReset == nil { return nil @@ -1860,13 +1913,18 @@ func (gs *LocalGovernanceStore) resetExpiredRateLimitFromSnapshot(ctx context.Co gs.LastDBUsagesRequestsRateLimits[resetRateLimit.ID] = 0 gs.LastDBUsagesRateLimitsRequestsMu.Unlock() } - gs.updateRateLimitReferences(ctx, resetRateLimit) + if refreshReferences { + gs.updateRateLimitReferences(ctx, resetRateLimit) + } return resetRateLimit } -// ResetExpiredRateLimitsInMemory performs background reset of expired rate limits for both provider-level and VK-level. +// ResetExpiredRateLimitsInMemory performs reset of expired rate limits for both provider-level and VK-level. // With no rateLimitIDs it scans every rate limit; with IDs it only checks those rate limits. -func (gs *LocalGovernanceStore) ResetExpiredRateLimitsInMemory(ctx context.Context, rateLimitIDs ...string) []*configstoreTables.TableRateLimit { +// refreshReferences controls whether embedded owner references (VK, team, customer) are updated +// after reset. Background/ticker callers pass true; request-time callers pass false to avoid +// an O(N) scan of all owners on the hot path. +func (gs *LocalGovernanceStore) ResetExpiredRateLimitsInMemory(ctx context.Context, refreshReferences bool, rateLimitIDs ...string) []*configstoreTables.TableRateLimit { now := time.Now() var resetRateLimits []*configstoreTables.TableRateLimit resetOne := func(value any) { @@ -1874,7 +1932,7 @@ func (gs *LocalGovernanceStore) ResetExpiredRateLimitsInMemory(ctx context.Conte if !ok || rateLimit == nil { return } - if resetRateLimit := gs.resetExpiredRateLimitFromSnapshot(ctx, rateLimit, now); resetRateLimit != nil { + if resetRateLimit := gs.resetExpiredRateLimitFromSnapshot(ctx, rateLimit, now, refreshReferences); resetRateLimit != nil { resetRateLimits = append(resetRateLimits, resetRateLimit) } } @@ -2341,6 +2399,7 @@ func (gs *LocalGovernanceStore) loadFromConfigMemory(ctx context.Context, config func (gs *LocalGovernanceStore) rebuildInMemoryStructures(ctx context.Context, customers []configstoreTables.TableCustomer, teams []configstoreTables.TableTeam, virtualKeys []configstoreTables.TableVirtualKey, budgets []configstoreTables.TableBudget, rateLimits []configstoreTables.TableRateLimit, modelConfigs []configstoreTables.TableModelConfig, providers []configstoreTables.TableProvider, routingRules []configstoreTables.TableRoutingRule) { // Clear existing data by creating new sync.Maps gs.virtualKeys = sync.Map{} + gs.virtualKeysByID = sync.Map{} gs.teams = sync.Map{} gs.customers = sync.Map{} gs.budgets = sync.Map{} @@ -2376,7 +2435,7 @@ func (gs *LocalGovernanceStore) rebuildInMemoryStructures(ctx context.Context, c // Build virtual keys map and track active VKs for i := range virtualKeys { vk := &virtualKeys[i] - gs.virtualKeys.Store(vk.Value, vk) + gs.storeVirtualKey(vk.Value.GetValue(), vk) } // Build model configs map. @@ -2872,7 +2931,7 @@ func (gs *LocalGovernanceStore) CreateVirtualKeyInMemory(ctx context.Context, vk } } - gs.virtualKeys.Store(clone.Value, &clone) + gs.storeVirtualKey(clone.Value.GetValue(), &clone) } // UpdateVirtualKeyInMemory updates an existing virtual key in the in-memory store (lock-free) @@ -2883,8 +2942,8 @@ func (gs *LocalGovernanceStore) UpdateVirtualKeyInMemory(ctx context.Context, vk // Do not update the current usage of the rate limit, as it will be updated by the usage tracker. // But update if max limit or reset duration changes. - existingVKKey := vk.Value - existingVKValue, exists := gs.virtualKeys.Load(vk.Value) + existingVKKey := vk.Value.GetValue() + existingVKValue, exists := gs.virtualKeys.Load(vk.Value.GetValue()) if exists && existingVKValue != nil { if existingVK, ok := existingVKValue.(*configstoreTables.TableVirtualKey); !ok || existingVK == nil || existingVK.ID != vk.ID { exists = false @@ -3047,10 +3106,10 @@ func (gs *LocalGovernanceStore) UpdateVirtualKeyInMemory(ctx context.Context, vk } } } - if existingVKKey != "" && existingVKKey != vk.Value { - gs.virtualKeys.Delete(existingVKKey) + if existingVKKey != "" && existingVKKey != vk.Value.GetValue() { + gs.deleteVirtualKeyByValue(existingVKKey) } - gs.virtualKeys.Store(vk.Value, &clone) + gs.storeVirtualKey(vk.Value.GetValue(), &clone) } else { gs.CreateVirtualKeyInMemory(ctx, vk) } @@ -3093,7 +3152,7 @@ func (gs *LocalGovernanceStore) DeleteVirtualKeyInMemory(ctx context.Context, vk } } - gs.virtualKeys.Delete(key) + gs.deleteVirtualKeyByValue(key.(string)) return false // stop iteration } return true // continue iteration @@ -3261,7 +3320,7 @@ func (gs *LocalGovernanceStore) DeleteTeamInMemory(ctx context.Context, teamID s clone := *vk clone.TeamID = nil clone.Team = nil - gs.virtualKeys.Store(key, &clone) + gs.storeVirtualKey(key.(string), &clone) } return true // continue iteration }) @@ -3377,7 +3436,7 @@ func (gs *LocalGovernanceStore) DeleteCustomerInMemory(ctx context.Context, cust clone := *vk clone.CustomerID = nil clone.Customer = nil - gs.virtualKeys.Store(key, &clone) + gs.storeVirtualKey(key.(string), &clone) } return true // continue iteration }) @@ -3634,7 +3693,7 @@ func (gs *LocalGovernanceStore) updateBudgetReferences(ctx context.Context, rese } } if needsUpdate { - gs.virtualKeys.Store(key, &clone) + gs.storeVirtualKey(key.(string), &clone) } return true // continue }) @@ -3704,7 +3763,7 @@ func (gs *LocalGovernanceStore) updateRateLimitReferences(ctx context.Context, r } if needsUpdate { - gs.virtualKeys.Store(key, &clone) + gs.storeVirtualKey(key.(string), &clone) } return true // continue }) diff --git a/plugins/governance/store_test.go b/plugins/governance/store_test.go index 256fcb7d0fe..620dfb4dbda 100644 --- a/plugins/governance/store_test.go +++ b/plugins/governance/store_test.go @@ -656,7 +656,7 @@ func TestGovernanceStore_UpdateVirtualKeyInMemory_RotatedValueRemovesOldLookup(t require.NoError(t, err) updated := *vk - updated.Value = "sk-bf-new" + updated.Value = *schemas.NewSecretVar("sk-bf-new") store.UpdateVirtualKeyInMemory(context.Background(), &updated, nil, nil, nil) oldVK, oldFound := store.GetVirtualKey(context.Background(), "sk-bf-old") @@ -738,7 +738,7 @@ func TestGovernanceStore_ResetExpiredRateLimits(t *testing.T) { require.NoError(t, err) // Reset expired rate limits - expiredRateLimits := store.ResetExpiredRateLimitsInMemory(context.Background()) + expiredRateLimits := store.ResetExpiredRateLimitsInMemory(context.Background(), true) err = store.ResetExpiredRateLimits(context.Background(), expiredRateLimits) assert.NoError(t, err, "Reset should succeed") @@ -773,7 +773,7 @@ func TestGovernanceStore_ResetExpiredBudgets(t *testing.T) { require.NoError(t, err) // Reset expired budgets - expiredBudgets := store.ResetExpiredBudgetsInMemory(context.Background()) + expiredBudgets := store.ResetExpiredBudgetsInMemory(context.Background(), true) err = store.ResetExpiredBudgets(context.Background(), expiredBudgets) assert.NoError(t, err, "Reset should succeed") diff --git a/plugins/governance/test_utils.go b/plugins/governance/test_utils.go index 441eba3922d..e8a335186e9 100644 --- a/plugins/governance/test_utils.go +++ b/plugins/governance/test_utils.go @@ -80,7 +80,7 @@ func (ml *MockLogger) LogHTTPRequest(level schemas.LogLevel, msg string) schemas func buildVirtualKey(id, value, name string, isActive bool) *configstoreTables.TableVirtualKey { return &configstoreTables.TableVirtualKey{ ID: id, - Value: value, + Value: *schemas.NewSecretVar(value), Name: name, IsActive: &isActive, } diff --git a/plugins/governance/tracker.go b/plugins/governance/tracker.go index 3ad76902768..b2deab70747 100644 --- a/plugins/governance/tracker.go +++ b/plugins/governance/tracker.go @@ -240,13 +240,13 @@ func (t *UsageTracker) resetWorker(ctx context.Context) { // resetExpiredCounters manages periodic resets of usage counters AND budgets using flexible durations func (t *UsageTracker) resetExpiredCounters(ctx context.Context) { // ==== PART 1: Reset Rate Limits ==== - resetRateLimits := t.store.ResetExpiredRateLimitsInMemory(ctx) + resetRateLimits := t.store.ResetExpiredRateLimitsInMemory(ctx, true) if err := t.store.ResetExpiredRateLimits(ctx, resetRateLimits); err != nil { t.logger.Error("failed to reset expired rate limits: %v", err) } // ==== PART 2: Reset Budgets ==== - resetBudgets := t.store.ResetExpiredBudgetsInMemory(ctx) + resetBudgets := t.store.ResetExpiredBudgetsInMemory(ctx, true) if err := t.store.ResetExpiredBudgets(ctx, resetBudgets); err != nil { t.logger.Error("failed to reset expired budgets: %v", err) } @@ -315,7 +315,7 @@ func (t *UsageTracker) PerformStartupResets(ctx context.Context) error { // Reuse the shared in-memory reset path so startup, ticker, and request-time // resets all apply the same LastDB baseline and reset-hook side effects. rateLimitResetStart := time.Now() - resetRateLimits := t.store.ResetExpiredRateLimitsInMemory(ctx) + resetRateLimits := t.store.ResetExpiredRateLimitsInMemory(ctx, true) t.logger.Info("[startup-timing] PerformStartupResets in-memory reset of %d rate limits took %v", len(resetRateLimits), time.Since(rateLimitResetStart)) if err := t.store.ResetExpiredRateLimits(ctx, resetRateLimits); err != nil { errs = append(errs, fmt.Sprintf("failed to reset expired rate limits: %s", err.Error())) @@ -323,7 +323,7 @@ func (t *UsageTracker) PerformStartupResets(ctx context.Context) error { // DB reset is also handled by this function budgetResetStart := time.Now() - resetBudgets := t.store.ResetExpiredBudgetsInMemory(ctx) + resetBudgets := t.store.ResetExpiredBudgetsInMemory(ctx, true) t.logger.Info("[startup-timing] PerformStartupResets in-memory reset of %d budgets took %v", len(resetBudgets), time.Since(budgetResetStart)) if err := t.store.ResetExpiredBudgets(ctx, resetBudgets); err != nil { errs = append(errs, fmt.Sprintf("failed to reset expired budgets: %s", err.Error())) diff --git a/plugins/governance/utils.go b/plugins/governance/utils.go index a6281d20d45..e19cd64c715 100644 --- a/plugins/governance/utils.go +++ b/plugins/governance/utils.go @@ -50,7 +50,8 @@ func IsModelRequiredForRequest(requestType schemas.RequestType) bool { // For these requests, we will only check for provider filtering // Cached content list/retrieve/update/delete target a resource name (cachedContents/{id}), // not a model, so they carry no model to filter on; only create binds a cache to a model. - if requestType == schemas.ListModelsRequest || requestType == schemas.MCPToolExecutionRequest || requestType == schemas.BatchCreateRequest || requestType == schemas.BatchListRequest || requestType == schemas.BatchRetrieveRequest || requestType == schemas.BatchCancelRequest || requestType == schemas.BatchResultsRequest || requestType == schemas.FileUploadRequest || requestType == schemas.FileListRequest || requestType == schemas.FileRetrieveRequest || requestType == schemas.FileDeleteRequest || requestType == schemas.FileContentRequest || requestType == schemas.ContainerCreateRequest || requestType == schemas.ContainerListRequest || requestType == schemas.ContainerRetrieveRequest || requestType == schemas.ContainerDeleteRequest || requestType == schemas.ContainerFileCreateRequest || requestType == schemas.ContainerFileListRequest || requestType == schemas.ContainerFileRetrieveRequest || requestType == schemas.ContainerFileContentRequest || requestType == schemas.ContainerFileDeleteRequest || requestType == schemas.CachedContentListRequest || requestType == schemas.CachedContentRetrieveRequest || requestType == schemas.CachedContentUpdateRequest || requestType == schemas.CachedContentDeleteRequest || requestType == schemas.VideoRetrieveRequest || requestType == schemas.VideoDownloadRequest || requestType == schemas.VideoListRequest || requestType == schemas.VideoDeleteRequest || requestType == schemas.VideoRemixRequest || requestType == schemas.PassthroughRequest || requestType == schemas.PassthroughStreamRequest { + // Responses retrieve/delete/cancel/input_items target a response_id, not a model. + if requestType == schemas.ListModelsRequest || requestType == schemas.MCPToolExecutionRequest || requestType == schemas.BatchCreateRequest || requestType == schemas.BatchListRequest || requestType == schemas.BatchRetrieveRequest || requestType == schemas.BatchCancelRequest || requestType == schemas.BatchResultsRequest || requestType == schemas.FileUploadRequest || requestType == schemas.FileListRequest || requestType == schemas.FileRetrieveRequest || requestType == schemas.FileDeleteRequest || requestType == schemas.FileContentRequest || requestType == schemas.ContainerCreateRequest || requestType == schemas.ContainerListRequest || requestType == schemas.ContainerRetrieveRequest || requestType == schemas.ContainerDeleteRequest || requestType == schemas.ContainerFileCreateRequest || requestType == schemas.ContainerFileListRequest || requestType == schemas.ContainerFileRetrieveRequest || requestType == schemas.ContainerFileContentRequest || requestType == schemas.ContainerFileDeleteRequest || requestType == schemas.CachedContentListRequest || requestType == schemas.CachedContentRetrieveRequest || requestType == schemas.CachedContentUpdateRequest || requestType == schemas.CachedContentDeleteRequest || requestType == schemas.ResponsesRetrieveRequest || requestType == schemas.ResponsesDeleteRequest || requestType == schemas.ResponsesCancelRequest || requestType == schemas.ResponsesInputItemsRequest || requestType == schemas.VideoRetrieveRequest || requestType == schemas.VideoDownloadRequest || requestType == schemas.VideoListRequest || requestType == schemas.VideoDeleteRequest || requestType == schemas.VideoRemixRequest || requestType == schemas.PassthroughRequest || requestType == schemas.PassthroughStreamRequest { return false } return true diff --git a/plugins/governance/version b/plugins/governance/version index fdd3be6df54..266146b87cb 100644 --- a/plugins/governance/version +++ b/plugins/governance/version @@ -1 +1 @@ -1.6.2 +1.6.3 diff --git a/plugins/jsonparser/changelog.md b/plugins/jsonparser/changelog.md index e69de29bb2d..a178a6f1d37 100644 --- a/plugins/jsonparser/changelog.md +++ b/plugins/jsonparser/changelog.md @@ -0,0 +1 @@ +- chore: upgraded core to v1.6.3 and framework to v1.4.3 diff --git a/plugins/jsonparser/version b/plugins/jsonparser/version index 32461d591e1..5b5dc42006b 100644 --- a/plugins/jsonparser/version +++ b/plugins/jsonparser/version @@ -1 +1 @@ -1.5.25 +1.5.26 diff --git a/plugins/logging/changelog.md b/plugins/logging/changelog.md index e69de29bb2d..1c8e2b1a222 100644 --- a/plugins/logging/changelog.md +++ b/plugins/logging/changelog.md @@ -0,0 +1,7 @@ +- feat: latency info on errors (#4867) +- feat: stream cost recalculation progress via SSE with batch processing (#4778) +- fix: sanitize `ErrorDetailsParsed` so raw payloads honor `disable_content_logging` (#4873, closes #4872) (thanks [@citrocat](https://github.com/citrocat)!) +- fix: sanitize error details on the log update path and remove redundant immediate error serialization (#4913) +- fix: cancelled state in logs (#4831, closes #3357) +- fix: empty tool call result insertion failures (#4925) +- chore: upgraded core to v1.6.3 and framework to v1.4.3 diff --git a/plugins/logging/go.mod b/plugins/logging/go.mod index 9b39e0a963a..fe2f89c4af4 100644 --- a/plugins/logging/go.mod +++ b/plugins/logging/go.mod @@ -6,6 +6,7 @@ require ( github.com/bytedance/sonic v1.15.1 github.com/maximhq/bifrost/core v1.6.2 github.com/maximhq/bifrost/framework v1.4.2 + github.com/stretchr/testify v1.11.1 ) require ( @@ -114,7 +115,6 @@ require ( github.com/rs/zerolog v1.34.0 // indirect github.com/spf13/cast v1.10.0 // indirect github.com/spiffe/go-spiffe/v2 v2.6.0 // indirect - github.com/stretchr/testify v1.11.1 // indirect github.com/tidwall/gjson v1.18.0 // indirect github.com/tidwall/match v1.1.1 // indirect github.com/tidwall/pretty v1.2.0 // indirect diff --git a/plugins/logging/main.go b/plugins/logging/main.go index f19ef3f78dc..dfc2982aa15 100644 --- a/plugins/logging/main.go +++ b/plugins/logging/main.go @@ -110,7 +110,13 @@ func applyLargePayloadPreviewsToEntry(ctx *schemas.BifrostContext, entry *logsto // sanitizeErrorForLogging returns a shallow copy of err with ExtraFields.RawRequest and // RawResponse cleared when raw-byte persistence is disabled, preventing raw bytes from -// leaking into entry.ErrorDetails via JSON serialization. +// leaking into the store via JSON serialization. +// +// Every assignment to ErrorDetailsParsed (Log and MCPToolLog alike) must go through this +// function: logstore's SerializeFields, which runs on every write path (BeforeCreate hook, +// hybrid store, rdb batch writes), serializes ErrorDetailsParsed into the error_details +// column and overwrites anything a caller put in ErrorDetails. Callers set only the +// sanitized ErrorDetailsParsed and leave the string serialization to SerializeFields. func sanitizeErrorForLogging(err *schemas.BifrostError, contentLoggingEnabled, shouldStoreRaw bool) *schemas.BifrostError { if err == nil { return nil @@ -259,6 +265,16 @@ type RecalculateCostResult struct { Remaining int64 `json:"remaining"` } +// RecalculateCostProgress represents a progress event from a cost backfill operation. +type RecalculateCostProgress struct { + TotalMatched int64 `json:"total_matched"` + Processed int `json:"processed"` + Updated int `json:"updated"` + Skipped int `json:"skipped"` + Remaining *int64 `json:"remaining,omitempty"` + Done bool `json:"done"` +} + // LogMessage represents a message in the logging queue type LogMessage struct { Operation LogOperation @@ -792,7 +808,7 @@ func (p *LoggerPlugin) PreLLMHook(ctx *schemas.BifrostContext, req *schemas.Bifr } initialData.RoutingEngineUsed = routingEngines - initialData.Status = "processing" + initialData.Status = logStatusProcessing // Store input data in pendingLogs for later combination with PostLLMHook output. // No DB write here - the write is deferred to PostLLMHook to halve total writes. @@ -804,7 +820,7 @@ func (p *LoggerPlugin) PreLLMHook(ctx *schemas.BifrostContext, req *schemas.Bifr RoutingEnginesUsed: routingEngines, InitialData: initialData, CreatedAt: time.Now(), - Status: "processing", + Status: logStatusProcessing, } // Seed LastActivity so the first idle-eviction check has a baseline even if no // PostLLMHook chunk has fired yet. @@ -896,7 +912,7 @@ func (p *LoggerPlugin) PostLLMHook(ctx *schemas.BifrostContext, result *schemas. entry := &logstore.Log{ ID: requestID, Provider: string(bifrostErr.ExtraFields.Provider), - Status: "error", + Status: logStatusForError(bifrostErr), Object: string(requestType), Stream: bifrost.IsStreamRequestType(requestType), Timestamp: time.Now().UTC(), @@ -911,10 +927,7 @@ func (p *LoggerPlugin) PostLLMHook(ctx *schemas.BifrostContext, result *schemas. } applyModelAlias(entry, originalModelRequested, resolvedModelUsed) applyResolvedAliasInfo(entry, resolvedKeyAlias) - if data, err := sonic.Marshal(sanitizeErrorForLogging(bifrostErr, contentLoggingEnabled, shouldStoreRaw)); err == nil { - entry.ErrorDetails = string(data) - } - entry.ErrorDetailsParsed = bifrostErr + entry.ErrorDetailsParsed = sanitizeErrorForLogging(bifrostErr, contentLoggingEnabled, shouldStoreRaw) if nodeID, _ := p.clusterNodeID.Load().(string); nodeID != "" { entry.ClusterNodeID = &nodeID } @@ -988,6 +1001,8 @@ func (p *LoggerPlugin) PostLLMHook(ctx *schemas.BifrostContext, result *schemas. if ef.CacheDebug != nil && ef.CacheDebug.CacheHit && ef.CacheDebug.CacheHitLatency != nil { latency = *ef.CacheDebug.CacheHitLatency } + } else if bifrostErr != nil { + latency = bifrostErr.ExtraFields.Latency } applyOutputFieldsToEntry(entry, selectedKeyID, selectedKeyName, virtualKeyID, virtualKeyName, routingRuleID, routingRuleName, selectedPromptID, selectedPromptName, selectedPromptVersion, teamID, teamName, customerID, customerName, userID, userName, businessUnitID, businessUnitName, numberOfRetries, latency, attemptTrail) applyResolvedAliasInfo(entry, resolvedKeyAlias) @@ -1027,7 +1042,7 @@ func (p *LoggerPlugin) PostLLMHook(ctx *schemas.BifrostContext, result *schemas. // Path A: Error with nil result if result == nil && bifrostErr != nil { - entry.Status = "error" + entry.Status = logStatusForError(bifrostErr) applyModelAlias(entry, originalModelRequested, resolvedModelUsed) if bifrost.IsStreamRequestType(requestType) { entry.Stream = true @@ -1046,13 +1061,7 @@ func (p *LoggerPlugin) PostLLMHook(ctx *schemas.BifrostContext, result *schemas. tracer.CleanupStreamAccumulator(traceID) } - // Serialize error details immediately since bifrostErr may be released - // back to the pool before the async batch writer processes this entry. - // Also set ErrorDetailsParsed for UI callback (JSON serialization uses this field). - if data, err := sonic.Marshal(sanitizeErrorForLogging(bifrostErr, contentLoggingEnabled, shouldStoreRaw)); err == nil { - entry.ErrorDetails = string(data) - } - entry.ErrorDetailsParsed = bifrostErr + entry.ErrorDetailsParsed = sanitizeErrorForLogging(bifrostErr, contentLoggingEnabled, shouldStoreRaw) if shouldStoreRaw && contentLoggingEnabled { if bifrostErr.ExtraFields.RawRequest != nil { rawReqBytes, err := sonic.Marshal(bifrostErr.ExtraFields.RawRequest) @@ -1090,13 +1099,10 @@ func (p *LoggerPlugin) PostLLMHook(ctx *schemas.BifrostContext, result *schemas. } if bifrostErr != nil { - entry.Status = "error" + entry.Status = logStatusForError(bifrostErr) entry.Stream = true applyModelAlias(entry, originalModelRequested, resolvedModelUsed) - if data, err := sonic.Marshal(sanitizeErrorForLogging(bifrostErr, contentLoggingEnabled, shouldStoreRaw)); err == nil { - entry.ErrorDetails = string(data) - } - entry.ErrorDetailsParsed = bifrostErr + entry.ErrorDetailsParsed = sanitizeErrorForLogging(bifrostErr, contentLoggingEnabled, shouldStoreRaw) // Backfill raw request/response on streaming-error path so cancellation/timeout // log entries still carry raw payloads when content logging + raw storage are // enabled. Mirrors the non-streaming Path A pattern at line 872. Prefer the @@ -1131,7 +1137,7 @@ func (p *LoggerPlugin) PostLLMHook(ctx *schemas.BifrostContext, result *schemas. } } else if streamResponse == nil { // tracer or traceID not available, or accumulator returned nil - still write what we have - entry.Status = "success" + entry.Status = logStatusSuccess entry.Stream = true applyModelAlias(entry, originalModelRequested, resolvedModelUsed) } else if isFinalChunk { @@ -1139,8 +1145,8 @@ func (p *LoggerPlugin) PostLLMHook(ctx *schemas.BifrostContext, result *schemas. entry.Stream = true p.applyStreamingOutputToEntry(entry, streamResponse, shouldStoreRaw, contentLoggingEnabled) } - if entry.ErrorDetails != "" || entry.ErrorDetailsParsed != nil { - entry.Status = "error" + if entry.ErrorDetailsParsed != nil { + entry.Status = logStatusForError(entry.ErrorDetailsParsed) } // Backfill passthrough status_code from response (streaming path) if result != nil && result.PassthroughResponse != nil { @@ -1152,7 +1158,7 @@ func (p *LoggerPlugin) PostLLMHook(ctx *schemas.BifrostContext, result *schemas. } // Flip status for passthrough error responses (4xx/5xx from provider) if isPassthroughErrorResponse(result) { - entry.Status = "error" + entry.Status = logStatusError } // Compute cost for streaming passthrough using StreamUsage set by the accumulator. if entry.Cost == nil && p.pricingManager != nil && result.PassthroughResponse.PassthroughUsage != nil { @@ -1173,22 +1179,16 @@ func (p *LoggerPlugin) PostLLMHook(ctx *schemas.BifrostContext, result *schemas. // Path C: Non-streaming response if bifrostErr != nil { - entry.Status = "error" + entry.Status = logStatusForError(bifrostErr) applyModelAlias(entry, originalModelRequested, resolvedModelUsed) - // Serialize error details immediately since bifrostErr may be released - // back to the pool before the async batch writer processes this entry. - // Also set ErrorDetailsParsed for UI callback (JSON serialization uses this field). - if data, err := sonic.Marshal(sanitizeErrorForLogging(bifrostErr, contentLoggingEnabled, shouldStoreRaw)); err == nil { - entry.ErrorDetails = string(data) - } - entry.ErrorDetailsParsed = bifrostErr + entry.ErrorDetailsParsed = sanitizeErrorForLogging(bifrostErr, contentLoggingEnabled, shouldStoreRaw) // Realtime turns that fail mid-stream still need their input transcript // surfaced — backfill from bifrostErr.ExtraFields.RawRequest if present. if requestType == schemas.RealtimeRequest { applyRealtimeRawRequestBackfill(entry, bifrostErr.ExtraFields.RawRequest, contentLoggingEnabled, shouldStoreRaw) } } else if result != nil { - entry.Status = "success" + entry.Status = logStatusSuccess extraFields := result.GetExtraFields() applyModelAlias(entry, extraFields.OriginalModelRequested, extraFields.ResolvedModelUsed) if requestType == schemas.RealtimeRequest { @@ -1198,7 +1198,7 @@ func (p *LoggerPlugin) PostLLMHook(ctx *schemas.BifrostContext, result *schemas. } // Flip status for passthrough error responses (4xx/5xx from provider) if isPassthroughErrorResponse(result) { - entry.Status = "error" + entry.Status = logStatusError } } applyLargePayloadPreviewsToEntry(ctx, entry, contentLoggingEnabled) @@ -1585,7 +1585,8 @@ func (p *LoggerPlugin) PostMCPHook(ctx *schemas.BifrostContext, resp *schemas.Bi if bifrostErr != nil { entry.Status = "error" - entry.ErrorDetailsParsed = bifrostErr + shouldStoreRaw, _ := ctx.Value(schemas.BifrostContextKeyShouldStoreRawInLogs).(bool) + entry.ErrorDetailsParsed = sanitizeErrorForLogging(bifrostErr, p.contentLoggingEnabled(ctx), shouldStoreRaw) } else if resp != nil { entry.Status = "success" if p.contentLoggingEnabled(ctx) { diff --git a/plugins/logging/operations.go b/plugins/logging/operations.go index a126d3cb568..936cdecd51e 100644 --- a/plugins/logging/operations.go +++ b/plugins/logging/operations.go @@ -16,6 +16,52 @@ import ( const realtimeMissingTranscriptText = "[Audio transcription unavailable]" +const ( + logStatusProcessing = "processing" + logStatusSuccess = "success" + logStatusError = "error" + logStatusCancelled = "cancelled" +) + +func logStatusForError(err *schemas.BifrostError) string { + if isCancelledLogError(err) { + return logStatusCancelled + } + return logStatusError +} + +func isCancelledLogError(err *schemas.BifrostError) bool { + if err == nil { + return false + } + if err.StatusCode != nil && *err.StatusCode == 499 { + return true + } + if err.Error == nil || err.Error.Type == nil { + return false + } + switch *err.Error.Type { + case schemas.RequestCancelled: + return true + case schemas.RequestTimedOut: + return isContextTimeoutLogError(err) + default: + return false + } +} + +func isContextTimeoutLogError(err *schemas.BifrostError) bool { + if err == nil || err.Error == nil { + return false + } + message := strings.ToLower(strings.TrimSpace(err.Error.Message)) + if message == "" || message == strings.ToLower(schemas.ErrProviderRequestTimedOut) { + return false + } + return strings.Contains(message, "by context") || + strings.Contains(message, "context deadline exceeded") +} + // insertInitialLogEntry creates a new log entry in the database using GORM func (p *LoggerPlugin) insertInitialLogEntry( ctx context.Context, @@ -33,7 +79,7 @@ func (p *LoggerPlugin) insertInitialLogEntry( Provider: data.Provider, Model: data.Model, FallbackIndex: fallbackIndex, - Status: "processing", + Status: logStatusProcessing, Stream: false, CreatedAt: timestamp, // Set parsed fields for serialization @@ -273,7 +319,8 @@ func (p *LoggerPlugin) updateLogEntry( } if data.ErrorDetails != nil { - tempEntry.ErrorDetailsParsed = data.ErrorDetails + shouldStoreRaw, _ := ctx.Value(schemas.BifrostContextKeyShouldStoreRawInLogs).(bool) + tempEntry.ErrorDetailsParsed = sanitizeErrorForLogging(data.ErrorDetails, contentLoggingEnabled, shouldStoreRaw) needsSerialization = true } @@ -339,15 +386,12 @@ func (p *LoggerPlugin) applyStreamingOutputToEntry(entry *logstore.Log, streamRe // Handle error case first if streamResponse.Data.ErrorDetails != nil { - entry.Status = "error" - // Serialize error details immediately to avoid use-after-free with pooled errors - if data, err := sonic.Marshal(streamResponse.Data.ErrorDetails); err == nil { - entry.ErrorDetails = string(data) - } + entry.Status = logStatusForError(streamResponse.Data.ErrorDetails) + entry.ErrorDetailsParsed = sanitizeErrorForLogging(streamResponse.Data.ErrorDetails, contentLoggingEnabled, shouldStoreRaw) latF := float64(streamResponse.Data.Latency) entry.Latency = &latF } else { - entry.Status = "success" + entry.Status = logStatusSuccess latF := float64(streamResponse.Data.Latency) entry.Latency = &latF } @@ -1199,8 +1243,16 @@ func (p *LoggerPlugin) extractUniqueMCPKeyPairs(logs []logstore.MCPToolLog, extr return result } -// RecalculateCosts recomputes cost for log entries that are missing cost values +// RecalculateCosts recomputes cost for all log entries that are missing cost values. +// The limit controls batch size, not the total number of rows processed. func (p *LoggerPlugin) RecalculateCosts(ctx context.Context, filters logstore.SearchFilters, limit int) (*RecalculateCostResult, error) { + return p.RecalculateCostsWithProgress(ctx, filters, limit, nil) +} + +// RecalculateCostsWithProgress recomputes cost for all log entries that are missing cost values +// and invokes progress after each batch. The limit controls batch size, not the +// total number of rows processed. +func (p *LoggerPlugin) RecalculateCostsWithProgress(ctx context.Context, filters logstore.SearchFilters, limit int, progress func(RecalculateCostProgress)) (*RecalculateCostResult, error) { if p.pricingManager == nil { return nil, fmt.Errorf("pricing manager is not configured") } @@ -1221,32 +1273,71 @@ func (p *LoggerPlugin) RecalculateCosts(ctx context.Context, filters logstore.Se Order: "asc", } - searchResult, err := p.store.SearchLogs(ctx, filters, pagination) - if err != nil { - return nil, fmt.Errorf("failed to search logs for cost recalculation: %w", err) - } + result := &RecalculateCostResult{} + seenInitialTotal := false + remainingOffset := 0 + processed := 0 - result := &RecalculateCostResult{ - TotalMatched: searchResult.Stats.TotalRequests, - } + for { + pagination.Offset = remainingOffset + searchResult, err := p.store.SearchLogs(ctx, filters, pagination) + if err != nil { + return nil, fmt.Errorf("failed to search logs for cost recalculation: %w", err) + } + if !seenInitialTotal { + result.TotalMatched = searchResult.Stats.TotalRequests + seenInitialTotal = true + } + if len(searchResult.Logs) == 0 { + break + } + processed += len(searchResult.Logs) - costUpdates := make(map[string]float64, len(searchResult.Logs)) + costUpdates := make(map[string]float64, len(searchResult.Logs)) + stillMissingInBatch := 0 - for _, logEntry := range searchResult.Logs { - cost, calcErr := p.calculateCostForLog(&logEntry) - if calcErr != nil { - result.Skipped++ - p.logger.Debug("skipping cost recalculation for log %s: %v", logEntry.ID, calcErr) - continue + for _, logEntry := range searchResult.Logs { + cost, calcErr := p.calculateCostForLog(&logEntry) + if calcErr != nil { + result.Skipped++ + stillMissingInBatch++ + p.logger.Debug("skipping cost recalculation for log %s: %v", logEntry.ID, calcErr) + continue + } + if cost <= 0 { + if isKnownZeroCostLog(&logEntry) { + costUpdates[logEntry.ID] = cost + } else { + result.Skipped++ + p.logger.Debug("skipping cost recalculation for log %s: resolved cost is zero", logEntry.ID) + } + // MissingCostOnly currently includes zero-cost rows, so advance past them + // whether they were skipped or updated to avoid recalculating forever. + stillMissingInBatch++ + continue + } + costUpdates[logEntry.ID] = cost } - costUpdates[logEntry.ID] = cost - } - if len(costUpdates) > 0 { - if err := p.store.BulkUpdateCost(ctx, costUpdates); err != nil { - return nil, fmt.Errorf("failed to bulk update costs: %w", err) + if len(costUpdates) > 0 { + if err := p.store.BulkUpdateCost(ctx, costUpdates); err != nil { + return nil, fmt.Errorf("failed to bulk update costs: %w", err) + } + result.Updated += len(costUpdates) + } + + remainingOffset += stillMissingInBatch + if progress != nil { + progress(RecalculateCostProgress{ + TotalMatched: result.TotalMatched, + Processed: processed, + Updated: result.Updated, + Skipped: result.Skipped, + }) + } + if len(searchResult.Logs) < limit { + break } - result.Updated = len(costUpdates) } // Re-count how many logs still match the missing-cost filter after updates @@ -1261,10 +1352,41 @@ func (p *LoggerPlugin) RecalculateCosts(ctx context.Context, filters logstore.Se } else { result.Remaining = remainingResult.Stats.TotalRequests } + if progress != nil { + remaining := result.Remaining + progress(RecalculateCostProgress{ + TotalMatched: result.TotalMatched, + Processed: processed, + Updated: result.Updated, + Skipped: result.Skipped, + Remaining: &remaining, + Done: true, + }) + } return result, nil } +func isKnownZeroCostLog(logEntry *logstore.Log) bool { + if logEntry == nil || logEntry.CacheDebugParsed == nil || !logEntry.CacheDebugParsed.CacheHit { + return false + } + return logEntry.CacheDebugParsed.HitType != nil && *logEntry.CacheDebugParsed.HitType == "direct" +} + +func normalizeLogRequestType(object string) schemas.RequestType { + switch object { + case "chat.completion": + return schemas.ChatCompletionRequest + case "chat.completion.chunk": + return schemas.ChatCompletionStreamRequest + case "response": + return schemas.ResponsesRequest + default: + return schemas.RequestType(object) + } +} + func (p *LoggerPlugin) calculateCostForLog(logEntry *logstore.Log) (float64, error) { if logEntry == nil { return 0, fmt.Errorf("log entry cannot be nil") @@ -1285,7 +1407,7 @@ func (p *LoggerPlugin) calculateCostForLog(logEntry *logstore.Log) (float64, err return 0, fmt.Errorf("token usage not available for log %s", logEntry.ID) } - requestType := schemas.RequestType(logEntry.Object) + requestType := normalizeLogRequestType(logEntry.Object) if requestType == "" && (cacheDebug == nil || !cacheDebug.CacheHit) { p.logger.Warn("skipping cost calculation for log %s: object type is empty (timestamp: %s)", logEntry.ID, logEntry.Timestamp) return 0, fmt.Errorf("object type is empty for log %s", logEntry.ID) diff --git a/plugins/logging/operations_test.go b/plugins/logging/operations_test.go index ab547e554b8..3f115e7840d 100644 --- a/plugins/logging/operations_test.go +++ b/plugins/logging/operations_test.go @@ -12,6 +12,7 @@ import ( "github.com/maximhq/bifrost/framework/logstore" "github.com/maximhq/bifrost/framework/modelcatalog" "github.com/maximhq/bifrost/framework/modelcatalog/datasheet" + "github.com/maximhq/bifrost/framework/streaming" ) type testLogger struct{} @@ -188,7 +189,7 @@ func TestPostLLMHookStreamingErrorPreservesHeaderMetadata(t *testing.T) { // TestPostLLMHookCancelledStreamLogsCost verifies #3357 at the logging layer: a // streaming request cancelled mid-flight (result==nil) whose error carries the // partial usage the provider already processed (BifrostError.ExtraFields.BilledUsage) -// must produce a log row with status="error", the consumed tokens, AND an +// must produce a log row with status="cancelled", the consumed tokens, AND an // accurate cost computed from the datasheet rates. func TestPostLLMHookCancelledStreamLogsCost(t *testing.T) { store := newTestStore(t) @@ -255,8 +256,8 @@ func TestPostLLMHookCancelledStreamLogsCost(t *testing.T) { if err != nil { t.Fatalf("FindByID() error = %v", err) } - if entry.Status != "error" { - t.Fatalf("expected error status, got %q", entry.Status) + if entry.Status != "cancelled" { + t.Fatalf("expected cancelled status, got %q", entry.Status) } if entry.TokenUsageParsed == nil { t.Fatalf("expected token usage recorded from BilledUsage on the cancel path") @@ -274,6 +275,171 @@ func TestPostLLMHookCancelledStreamLogsCost(t *testing.T) { } } +func TestPostLLMHookContextTimeoutLogsCancelledStatus(t *testing.T) { + store := newTestStore(t) + plugin, err := Init(context.Background(), &Config{}, testLogger{}, store, nil, nil) + if err != nil { + t.Fatalf("Init() error = %v", err) + } + + ctx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline) + ctx.SetValue(schemas.BifrostContextKeyRequestID, "req-context-timeout") + + req := &schemas.BifrostRequest{ + RequestType: schemas.ChatCompletionRequest, + ChatRequest: &schemas.BifrostChatRequest{ + Provider: schemas.OpenAI, + Model: "gpt-4o", + Params: &schemas.ChatParameters{}, + }, + } + if _, _, err = plugin.PreLLMHook(ctx, req); err != nil { + t.Fatalf("PreLLMHook() error = %v", err) + } + + statusCode := 504 + bifrostErr := &schemas.BifrostError{ + IsBifrostError: true, + StatusCode: &statusCode, + Error: &schemas.ErrorField{ + Message: "Request timed out by context: context deadline exceeded", + Type: schemas.Ptr(schemas.RequestTimedOut), + }, + ExtraFields: schemas.BifrostErrorExtraFields{ + RequestType: schemas.ChatCompletionRequest, + Provider: schemas.OpenAI, + OriginalModelRequested: "gpt-4o", + ResolvedModelUsed: "gpt-4o", + }, + } + if _, _, err = plugin.PostLLMHook(ctx, nil, bifrostErr); err != nil { + t.Fatalf("PostLLMHook() error = %v", err) + } + if err := plugin.Cleanup(); err != nil { + t.Fatalf("Cleanup() error = %v", err) + } + + entry, err := store.FindByID(context.Background(), "req-context-timeout") + if err != nil { + t.Fatalf("FindByID() error = %v", err) + } + if entry.Status != "cancelled" { + t.Fatalf("expected cancelled status, got %q", entry.Status) + } +} + +func TestPostLLMHookProviderTimeoutRemainsErrorStatus(t *testing.T) { + store := newTestStore(t) + plugin, err := Init(context.Background(), &Config{}, testLogger{}, store, nil, nil) + if err != nil { + t.Fatalf("Init() error = %v", err) + } + + ctx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline) + ctx.SetValue(schemas.BifrostContextKeyRequestID, "req-provider-timeout") + + req := &schemas.BifrostRequest{ + RequestType: schemas.ChatCompletionRequest, + ChatRequest: &schemas.BifrostChatRequest{ + Provider: schemas.OpenAI, + Model: "gpt-4o", + Params: &schemas.ChatParameters{}, + }, + } + if _, _, err = plugin.PreLLMHook(ctx, req); err != nil { + t.Fatalf("PreLLMHook() error = %v", err) + } + + statusCode := 504 + bifrostErr := &schemas.BifrostError{ + IsBifrostError: true, + StatusCode: &statusCode, + Error: &schemas.ErrorField{ + Message: schemas.ErrProviderRequestTimedOut, + Type: schemas.Ptr(schemas.RequestTimedOut), + }, + ExtraFields: schemas.BifrostErrorExtraFields{ + RequestType: schemas.ChatCompletionRequest, + Provider: schemas.OpenAI, + OriginalModelRequested: "gpt-4o", + ResolvedModelUsed: "gpt-4o", + }, + } + if _, _, err = plugin.PostLLMHook(ctx, nil, bifrostErr); err != nil { + t.Fatalf("PostLLMHook() error = %v", err) + } + if err := plugin.Cleanup(); err != nil { + t.Fatalf("Cleanup() error = %v", err) + } + + entry, err := store.FindByID(context.Background(), "req-provider-timeout") + if err != nil { + t.Fatalf("FindByID() error = %v", err) + } + if entry.Status != "error" { + t.Fatalf("expected error status, got %q", entry.Status) + } +} + +func TestLogStatusForErrorDoesNotTreatGenericDeadlineMessageAsCancelled(t *testing.T) { + statusCode := 504 + bifrostErr := &schemas.BifrostError{ + IsBifrostError: true, + StatusCode: &statusCode, + Error: &schemas.ErrorField{ + Message: "provider request hit project deadline exceeded limit", + Type: schemas.Ptr(schemas.RequestTimedOut), + }, + } + + if got := logStatusForError(bifrostErr); got != "error" { + t.Fatalf("expected generic provider deadline message to remain error, got %q", got) + } +} + +func TestApplyStreamingOutputToEntryPreservesAccumulatorCancelledStatus(t *testing.T) { + plugin := &LoggerPlugin{} + entry := &logstore.Log{} + statusCode := 499 + streamResponse := &streaming.ProcessedStreamResponse{ + Data: &streaming.AccumulatedData{ + ErrorDetails: &schemas.BifrostError{ + IsBifrostError: true, + StatusCode: &statusCode, + Error: &schemas.ErrorField{ + Message: "Request cancelled: client disconnected", + Type: schemas.Ptr(schemas.RequestCancelled), + }, + }, + Latency: 42, + }, + } + + plugin.applyStreamingOutputToEntry(entry, streamResponse, false, true) + if entry.Status != "cancelled" { + t.Fatalf("expected initial cancelled status, got %q", entry.Status) + } + if entry.ErrorDetailsParsed == nil { + t.Fatalf("expected parsed error details to be set") + } + if err := entry.SerializeFields(); err != nil { + t.Fatalf("SerializeFields() error: %v", err) + } + if entry.ErrorDetails == "" { + t.Fatalf("expected serialized error details after SerializeFields") + } + + // Match the downstream Path B re-derivation in PostLLMHook. If + // ErrorDetailsParsed is missing, this collapses accumulator-originated + // cancellations back to "error". + if entry.ErrorDetailsParsed != nil { + entry.Status = logStatusForError(entry.ErrorDetailsParsed) + } + if entry.Status != "cancelled" { + t.Fatalf("expected downstream reclassification to preserve cancelled status, got %q", entry.Status) + } +} + // newTestPricingManager builds a ModelCatalog backed by the committed pricing // testdata via an offline file:// URL (no network). func newTestPricingManager(t *testing.T) *modelcatalog.ModelCatalog { @@ -360,6 +526,186 @@ func TestApplyErrorBillingFromBilledUsage_FillsTokensAndCostWhenUnparsed(t *test } } +func TestRecalculateCostsProcessesAllBatches(t *testing.T) { + store := newTestStore(t) + plugin := &LoggerPlugin{ + store: store, + pricingManager: newTestPricingManager(t), + logger: testLogger{}, + } + + now := time.Now().UTC() + for i := range 5 { + if err := store.Create(context.Background(), testRecalculateCostLog( + "req-recalc-batch-"+string(rune('a'+i)), + now.Add(time.Duration(i)*time.Second), + 100+i, + 50, + )); err != nil { + t.Fatalf("Create() error = %v", err) + } + } + + result, err := plugin.RecalculateCosts(context.Background(), logstore.SearchFilters{}, 2) + if err != nil { + t.Fatalf("RecalculateCosts() error = %v", err) + } + if result.TotalMatched != 5 || result.Updated != 5 || result.Skipped != 0 || result.Remaining != 0 { + t.Fatalf("unexpected result: %+v", result) + } + + for _, id := range []string{"req-recalc-batch-a", "req-recalc-batch-b", "req-recalc-batch-c", "req-recalc-batch-d", "req-recalc-batch-e"} { + entry, err := store.FindByID(context.Background(), id) + if err != nil { + t.Fatalf("FindByID(%s) error = %v", id, err) + } + if entry.Cost == nil || *entry.Cost <= 0 { + t.Fatalf("expected positive cost for %s, got %v", id, entry.Cost) + } + } +} + +func TestRecalculateCostsSkipsUnresolvableRowsAndContinues(t *testing.T) { + store := newTestStore(t) + plugin := &LoggerPlugin{ + store: store, + pricingManager: newTestPricingManager(t), + logger: testLogger{}, + } + + now := time.Now().UTC() + if err := store.Create(context.Background(), &logstore.Log{ + ID: "req-recalc-skip", + Timestamp: now, + Object: string(schemas.ChatCompletionRequest), + Provider: string(schemas.OpenAI), + Model: "gpt-4o", + Status: "success", + }); err != nil { + t.Fatalf("Create() error = %v", err) + } + + for i := range 3 { + if err := store.Create(context.Background(), testRecalculateCostLog( + "req-recalc-continue-"+string(rune('a'+i)), + now.Add(time.Duration(i+1)*time.Second), + 100+i, + 50, + )); err != nil { + t.Fatalf("Create() error = %v", err) + } + } + + result, err := plugin.RecalculateCosts(context.Background(), logstore.SearchFilters{}, 2) + if err != nil { + t.Fatalf("RecalculateCosts() error = %v", err) + } + if result.TotalMatched != 4 || result.Updated != 3 || result.Skipped != 1 || result.Remaining != 1 { + t.Fatalf("unexpected result: %+v", result) + } + + skipped, err := store.FindByID(context.Background(), "req-recalc-skip") + if err != nil { + t.Fatalf("FindByID(req-recalc-skip) error = %v", err) + } + if skipped.Cost != nil { + t.Fatalf("expected skipped row cost to remain nil, got %v", skipped.Cost) + } + for _, id := range []string{"req-recalc-continue-a", "req-recalc-continue-b", "req-recalc-continue-c"} { + entry, err := store.FindByID(context.Background(), id) + if err != nil { + t.Fatalf("FindByID(%s) error = %v", id, err) + } + if entry.Cost == nil || *entry.Cost <= 0 { + t.Fatalf("expected positive cost for %s, got %v", id, entry.Cost) + } + } +} + +func TestRecalculateCostsDoesNotWriteZeroForUnresolvedPricing(t *testing.T) { + store := newTestStore(t) + plugin := &LoggerPlugin{ + store: store, + pricingManager: newTestPricingManager(t), + logger: testLogger{}, + } + + entry := testRecalculateCostLog("req-recalc-unpriced", time.Now().UTC(), 100, 50) + entry.Model = "unknown-model" + if err := store.Create(context.Background(), entry); err != nil { + t.Fatalf("Create() error = %v", err) + } + + result, err := plugin.RecalculateCosts(context.Background(), logstore.SearchFilters{}, 2) + if err != nil { + t.Fatalf("RecalculateCosts() error = %v", err) + } + if result.TotalMatched != 1 || result.Updated != 0 || result.Skipped != 1 || result.Remaining != 1 { + t.Fatalf("unexpected result: %+v", result) + } + + logEntry, err := store.FindByID(context.Background(), "req-recalc-unpriced") + if err != nil { + t.Fatalf("FindByID() error = %v", err) + } + if logEntry.Cost != nil { + t.Fatalf("expected unresolved pricing to leave cost nil, got %v", *logEntry.Cost) + } +} + +func TestRecalculateCostsNormalizesProviderObjectForPricing(t *testing.T) { + store := newTestStore(t) + plugin := &LoggerPlugin{ + store: store, + pricingManager: newTestPricingManager(t), + logger: testLogger{}, + } + + entry := testRecalculateCostLog("req-recalc-provider-object", time.Now().UTC(), 100, 50) + entry.Object = "chat.completion" + zeroCost := 0.0 + entry.Cost = &zeroCost + if err := store.Create(context.Background(), entry); err != nil { + t.Fatalf("Create() error = %v", err) + } + + result, err := plugin.RecalculateCosts(context.Background(), logstore.SearchFilters{}, 2) + if err != nil { + t.Fatalf("RecalculateCosts() error = %v", err) + } + if result.TotalMatched != 1 || result.Updated != 1 || result.Skipped != 0 || result.Remaining != 0 { + t.Fatalf("unexpected result: %+v", result) + } + + logEntry, err := store.FindByID(context.Background(), "req-recalc-provider-object") + if err != nil { + t.Fatalf("FindByID() error = %v", err) + } + if logEntry.Cost == nil || *logEntry.Cost <= 0 { + t.Fatalf("expected legacy provider object to recalculate positive cost, got %v", logEntry.Cost) + } +} + +func testRecalculateCostLog(id string, timestamp time.Time, promptTokens int, completionTokens int) *logstore.Log { + totalTokens := promptTokens + completionTokens + return &logstore.Log{ + ID: id, + Timestamp: timestamp, + Object: string(schemas.ChatCompletionRequest), + Provider: string(schemas.OpenAI), + Model: "gpt-4o", + Status: "success", + PromptTokens: promptTokens, + CompletionTokens: completionTokens, + TotalTokens: totalTokens, + TokenUsageParsed: &schemas.BifrostLLMUsage{ + PromptTokens: promptTokens, + CompletionTokens: completionTokens, + TotalTokens: totalTokens, + }, + } +} + func TestBuildInitialLogEntryPreservesMetadata(t *testing.T) { metadata := map[string]any{"tenant": "acme"} entry := buildInitialLogEntry(&PendingLogData{ diff --git a/plugins/logging/sanitize_test.go b/plugins/logging/sanitize_test.go new file mode 100644 index 00000000000..f33c47d0c44 --- /dev/null +++ b/plugins/logging/sanitize_test.go @@ -0,0 +1,149 @@ +package logging + +import ( + "context" + "strings" + "testing" + "time" + + "github.com/maximhq/bifrost/core/schemas" + "github.com/maximhq/bifrost/framework/logstore" +) + +func errorWithRawPayloads() *schemas.BifrostError { + return &schemas.BifrostError{ + IsBifrostError: false, + Error: &schemas.ErrorField{Message: "provider rejected request"}, + ExtraFields: schemas.BifrostErrorExtraFields{ + RawRequest: map[string]any{"messages": "RAW_REQUEST_MARKER"}, + RawResponse: map[string]any{"body": "RAW_RESPONSE_MARKER"}, + }, + } +} + +// Regression test: logstore's SerializeFields serializes ErrorDetailsParsed +// into ErrorDetails on write. If the parsed field holds the unsanitized error, +// raw request/response payloads reach the store even when content logging is +// disabled. +func TestSanitizedErrorDetailsSurviveSerializeFields(t *testing.T) { + entry := &logstore.Log{ID: "req-1"} + entry.ErrorDetailsParsed = sanitizeErrorForLogging(errorWithRawPayloads(), false, false) + + if entry.ErrorDetailsParsed == nil { + t.Fatal("ErrorDetailsParsed should be set") + } + if entry.ErrorDetailsParsed.ExtraFields.RawRequest != nil || + entry.ErrorDetailsParsed.ExtraFields.RawResponse != nil { + t.Error("ErrorDetailsParsed should not retain raw payloads when content logging is disabled") + } + + // Simulate the DB write path (BeforeCreate calls SerializeFields). + if err := entry.SerializeFields(); err != nil { + t.Fatalf("SerializeFields() error: %v", err) + } + if strings.Contains(entry.ErrorDetails, "RAW_REQUEST_MARKER") || + strings.Contains(entry.ErrorDetails, "RAW_RESPONSE_MARKER") { + t.Error("serialized ErrorDetails must not contain raw payloads when content logging is disabled") + } + if !strings.Contains(entry.ErrorDetails, "provider rejected request") { + t.Error("serialized ErrorDetails should still contain the error message") + } +} + +// When content logging and raw storage are both enabled, raw payloads are +// intentionally preserved. +func TestRawErrorDetailsPreservedWhenEnabled(t *testing.T) { + entry := &logstore.Log{ID: "req-2"} + entry.ErrorDetailsParsed = sanitizeErrorForLogging(errorWithRawPayloads(), true, true) + + if entry.ErrorDetailsParsed == nil { + t.Fatal("ErrorDetailsParsed should be set") + } + if entry.ErrorDetailsParsed.ExtraFields.RawRequest == nil { + t.Error("raw payloads should be preserved when content logging and raw storage are enabled") + } + if err := entry.SerializeFields(); err != nil { + t.Fatalf("SerializeFields() error: %v", err) + } + if !strings.Contains(entry.ErrorDetails, "RAW_REQUEST_MARKER") { + t.Error("serialized ErrorDetails should contain raw payloads when explicitly enabled") + } +} + +func TestSanitizeErrorForLoggingNilError(t *testing.T) { + entry := &logstore.Log{ID: "req-3"} + entry.ErrorDetailsParsed = sanitizeErrorForLogging(nil, false, false) + if entry.ErrorDetailsParsed != nil { + t.Error("nil error should leave ErrorDetailsParsed nil") + } + if err := entry.SerializeFields(); err != nil { + t.Fatalf("SerializeFields() error: %v", err) + } + if entry.ErrorDetails != "" { + t.Error("nil error should leave ErrorDetails empty") + } +} + +// MCPToolLog counterpart: same sanitization semantics through its own +// SerializeFields. +func TestSanitizedMCPErrorDetailsSurviveSerializeFields(t *testing.T) { + entry := &logstore.MCPToolLog{ID: "mcp-1"} + entry.ErrorDetailsParsed = sanitizeErrorForLogging(errorWithRawPayloads(), false, false) + + if entry.ErrorDetailsParsed == nil { + t.Fatal("ErrorDetailsParsed should be set") + } + if entry.ErrorDetailsParsed.ExtraFields.RawRequest != nil || + entry.ErrorDetailsParsed.ExtraFields.RawResponse != nil { + t.Error("ErrorDetailsParsed should not retain raw payloads when content logging is disabled") + } + if err := entry.SerializeFields(); err != nil { + t.Fatalf("SerializeFields() error: %v", err) + } + if strings.Contains(entry.ErrorDetails, "RAW_REQUEST_MARKER") { + t.Error("serialized ErrorDetails must not contain raw payloads when content logging is disabled") + } + if !strings.Contains(entry.ErrorDetails, "provider rejected request") { + t.Error("serialized ErrorDetails should still contain the error message") + } +} + +// Update-path regression: updateLogEntry must sanitize UpdateLogData.ErrorDetails +// before SerializeFields copies it into the error_details column update. +func TestUpdateLogEntrySanitizesErrorDetails(t *testing.T) { + store := newTestStore(t) + plugin := &LoggerPlugin{ + store: store, + logger: testLogger{}, + } + + requestID := "req-err-update" + initial := &InitialLogData{ + Object: "chat_completion", + Provider: "openai", + Model: "gpt-4o-mini", + } + if err := plugin.insertInitialLogEntry(context.Background(), requestID, "", time.Now().UTC(), 0, nil, initial); err != nil { + t.Fatalf("insertInitialLogEntry() error = %v", err) + } + + update := &UpdateLogData{ + Status: "error", + ErrorDetails: errorWithRawPayloads(), + } + if err := plugin.updateLogEntry(context.Background(), requestID, "", "", 10, "", "", "", "", 0, nil, "", update, false); err != nil { + t.Fatalf("updateLogEntry() error = %v", err) + } + + logEntry, err := store.FindByID(context.Background(), requestID) + if err != nil { + t.Fatalf("FindByID() error = %v", err) + } + if strings.Contains(logEntry.ErrorDetails, "RAW_REQUEST_MARKER") || + strings.Contains(logEntry.ErrorDetails, "RAW_RESPONSE_MARKER") { + t.Errorf("stored error_details must not contain raw payloads when content logging is disabled, got %q", logEntry.ErrorDetails) + } + if !strings.Contains(logEntry.ErrorDetails, "provider rejected request") { + t.Errorf("stored error_details should still contain the error message, got %q", logEntry.ErrorDetails) + } +} diff --git a/plugins/logging/strip.go b/plugins/logging/strip.go new file mode 100644 index 00000000000..79c742ff9c6 --- /dev/null +++ b/plugins/logging/strip.go @@ -0,0 +1,140 @@ +package logging + +import ( + "math" + "reflect" + + "github.com/bytedance/sonic" +) + +// stripUnserializablePayloads walks any value with reflection and zeroes out +// only the nested values that fail JSON serialization, preserving the rest. +// Subtrees that marshal cleanly are skipped whole; a value that still fails +// after sanitization (e.g. broken unexported state) is zeroed entirely by its +// parent. Pass a pointer (or map/slice) so repairs are visible to the caller; +// a plain struct value cannot be mutated through reflection. +func stripUnserializablePayloads(v any) { + if v == nil || marshals(v) { + return + } + sanitize(reflect.ValueOf(v), make(map[uintptr]bool), 0) +} + +// maxSanitizeDepth bounds recursion on pathologically nested payloads. +const maxSanitizeDepth = 64 + +// sanitize recursively zeroes out values that fail JSON marshaling. Subtrees +// that marshal cleanly are skipped whole, so the walk cost is bounded by the +// broken paths rather than the payload size. Values that cannot be repaired +// in place are zeroed by the parent via the post-recursion marshal check. +func sanitize(v reflect.Value, visited map[uintptr]bool, depth int) { + if depth > maxSanitizeDepth || !v.IsValid() { + return + } + switch v.Kind() { + case reflect.Interface: + if v.IsNil() || marshals(v.Interface()) { + return + } + // Interface contents are not addressable; sanitize a copy and + // write it back. + inner := v.Elem() + tmp := reflect.New(inner.Type()).Elem() + tmp.Set(inner) + sanitize(tmp, visited, depth+1) + if !v.CanSet() { + return + } + if marshals(tmp.Interface()) { + v.Set(tmp) + } else { + v.Set(reflect.Zero(v.Type())) + } + case reflect.Pointer: + if v.IsNil() { + return + } + ptr := v.Pointer() + if visited[ptr] { + // Reference cycle: break it by nilling the back-edge. + if v.CanSet() { + v.Set(reflect.Zero(v.Type())) + } + return + } + visited[ptr] = true + defer delete(visited, ptr) + if v.CanInterface() && marshals(v.Interface()) { + return + } + sanitize(v.Elem(), visited, depth+1) + if v.CanSet() && v.CanInterface() && !marshals(v.Interface()) { + v.Set(reflect.Zero(v.Type())) + } + case reflect.Map: + if v.IsNil() { + return + } + for _, k := range v.MapKeys() { + mv := v.MapIndex(k) + if !mv.CanInterface() || marshals(mv.Interface()) { + continue + } + // Map values are not addressable; sanitize a copy and + // store it back. + tmp := reflect.New(mv.Type()).Elem() + tmp.Set(mv) + sanitize(tmp, visited, depth+1) + if marshals(tmp.Interface()) { + v.SetMapIndex(k, tmp) + } else { + v.SetMapIndex(k, reflect.Zero(mv.Type())) + } + } + case reflect.Slice, reflect.Array: + for i := 0; i < v.Len(); i++ { + ev := v.Index(i) + if !ev.CanInterface() || marshals(ev.Interface()) { + continue + } + sanitize(ev, visited, depth+1) + if ev.CanSet() && ev.CanInterface() && !marshals(ev.Interface()) { + ev.Set(reflect.Zero(ev.Type())) + } + } + case reflect.Struct: + t := v.Type() + for i := 0; i < v.NumField(); i++ { + if !t.Field(i).IsExported() { + // JSON marshaling ignores unexported fields. + continue + } + fv := v.Field(i) + if !fv.CanInterface() || marshals(fv.Interface()) { + continue + } + sanitize(fv, visited, depth+1) + if fv.CanSet() && fv.CanInterface() && !marshals(fv.Interface()) { + fv.Set(reflect.Zero(fv.Type())) + } + } + case reflect.Float32, reflect.Float64: + f := v.Float() + if v.CanSet() && (math.IsNaN(f) || math.IsInf(f, 0)) { + v.SetFloat(0) + } + case reflect.Chan, reflect.Func, reflect.UnsafePointer, + reflect.Complex64, reflect.Complex128: + // Never JSON-serializable; the parent zeroes these via its + // post-recursion marshal check. + } +} + +// marshals reports whether v serializes cleanly to JSON; nil values trivially do. +func marshals(v any) bool { + if v == nil { + return true + } + _, err := sonic.Marshal(v) + return err == nil +} diff --git a/plugins/logging/strip_test.go b/plugins/logging/strip_test.go new file mode 100644 index 00000000000..327347d48bc --- /dev/null +++ b/plugins/logging/strip_test.go @@ -0,0 +1,148 @@ +package logging + +import ( + "errors" + "math" + "testing" + + "github.com/maximhq/bifrost/framework/logstore" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// failingMarshaler always errors from MarshalJSON, simulating a payload type +// with broken custom serialization. +type failingMarshaler struct{} + +func (failingMarshaler) MarshalJSON() ([]byte, error) { + return nil, errors.New("boom") +} + +type cyclicNode struct { + Name string `json:"name"` + Next *cyclicNode `json:"next,omitempty"` +} + +func TestStripUnserializablePayloadsNilAndClean(t *testing.T) { + // Nil must not panic. + stripUnserializablePayloads(nil) + + // Clean values must be left untouched. + clean := map[string]interface{}{"a": 1, "b": []string{"x"}} + stripUnserializablePayloads(clean) + assert.Equal(t, 1, clean["a"]) + assert.Equal(t, []string{"x"}, clean["b"]) +} + +func TestStripUnserializablePayloadsMapLeaf(t *testing.T) { + m := map[string]interface{}{ + "good": "keep-me", + "bad": make(chan int), + "nested": map[string]interface{}{ + "fn": func() {}, + "kept": 42, + }, + } + stripUnserializablePayloads(m) + + assert.Equal(t, "keep-me", m["good"]) + assert.Nil(t, m["bad"]) + nested, ok := m["nested"].(map[string]interface{}) + require.True(t, ok) + assert.Nil(t, nested["fn"]) + assert.Equal(t, 42, nested["kept"]) + assert.True(t, marshals(m)) +} + +func TestStripUnserializablePayloadsSliceElement(t *testing.T) { + s := []interface{}{"first", make(chan int), "third"} + stripUnserializablePayloads(s) + + assert.Equal(t, "first", s[0]) + assert.Nil(t, s[1]) + assert.Equal(t, "third", s[2]) + assert.True(t, marshals(s)) +} + +func TestStripUnserializablePayloadsNaNAndInf(t *testing.T) { + m := map[string]interface{}{ + "nan": math.NaN(), + "inf": math.Inf(1), + "cost": 1.25, + } + if marshals(m) { + t.Skip("sonic accepts NaN/Inf in this configuration; nothing to strip") + } + stripUnserializablePayloads(m) + + assert.Equal(t, float64(0), m["nan"]) + assert.Equal(t, float64(0), m["inf"]) + assert.Equal(t, 1.25, m["cost"]) + assert.True(t, marshals(m)) +} + +func TestStripUnserializablePayloadsFailingMarshaler(t *testing.T) { + m := map[string]interface{}{ + "broken": failingMarshaler{}, + "kept": "still-here", + } + stripUnserializablePayloads(m) + + assert.Nil(t, m["broken"]) + assert.Equal(t, "still-here", m["kept"]) + assert.True(t, marshals(m)) +} + +func TestStripUnserializablePayloadsCycle(t *testing.T) { + a := &cyclicNode{Name: "a"} + b := &cyclicNode{Name: "b", Next: a} + a.Next = b + if marshals(a) { + t.Skip("sonic tolerates reference cycles in this configuration") + } + stripUnserializablePayloads(a) + + assert.True(t, marshals(a)) + assert.Equal(t, "a", a.Name) +} + +func TestStripUnserializablePayloadsStructField(t *testing.T) { + type payload struct { + Kept string `json:"kept"` + Params map[string]interface{} `json:"params"` + } + p := &payload{ + Kept: "scalar", + Params: map[string]interface{}{"ch": make(chan int), "ok": true}, + } + stripUnserializablePayloads(p) + + assert.Equal(t, "scalar", p.Kept) + assert.Nil(t, p.Params["ch"]) + assert.Equal(t, true, p.Params["ok"]) + assert.True(t, marshals(p)) +} + +func TestStripUnserializablePayloadsLogEntry(t *testing.T) { + entry := &logstore.Log{ + ID: "log-1", + Model: "gpt-4o", + Status: "success", + ParamsParsed: map[string]interface{}{ + "temperature": 0.7, + "broken": make(chan int), + }, + } + stripUnserializablePayloads(entry) + + // Scalar columns untouched. + assert.Equal(t, "log-1", entry.ID) + assert.Equal(t, "gpt-4o", entry.Model) + assert.Equal(t, "success", entry.Status) + // Only the broken nested value is cleared; the rest of params survives. + params, ok := entry.ParamsParsed.(map[string]interface{}) + require.True(t, ok) + assert.Equal(t, 0.7, params["temperature"]) + assert.Nil(t, params["broken"]) + assert.True(t, marshals(entry)) +} diff --git a/plugins/logging/utils.go b/plugins/logging/utils.go index 660e7baf086..ee493d23d24 100644 --- a/plugins/logging/utils.go +++ b/plugins/logging/utils.go @@ -123,6 +123,8 @@ type LogManager interface { // RecalculateCosts recomputes missing costs for logs matching the filters RecalculateCosts(ctx context.Context, filters *logstore.SearchFilters, limit int) (*RecalculateCostResult, error) + // RecalculateCostsWithProgress recomputes missing costs and emits batch progress updates + RecalculateCostsWithProgress(ctx context.Context, filters *logstore.SearchFilters, limit int, progress func(RecalculateCostProgress)) (*RecalculateCostResult, error) // MCP Tool Log methods // GetMCPToolLog retrieves a single MCP tool log entry by ID. @@ -369,6 +371,13 @@ func (p *PluginLogManager) RecalculateCosts(ctx context.Context, filters *logsto return p.plugin.RecalculateCosts(ctx, *filters, limit) } +func (p *PluginLogManager) RecalculateCostsWithProgress(ctx context.Context, filters *logstore.SearchFilters, limit int, progress func(RecalculateCostProgress)) (*RecalculateCostResult, error) { + if filters == nil { + return nil, fmt.Errorf("filters cannot be nil") + } + return p.plugin.RecalculateCostsWithProgress(ctx, *filters, limit, progress) +} + // GetMCPToolLog retrieves a single MCP tool log entry by ID. func (p *PluginLogManager) GetMCPToolLog(ctx context.Context, id string) (*logstore.MCPToolLog, error) { if p.plugin == nil || p.plugin.store == nil { diff --git a/plugins/logging/version b/plugins/logging/version index 32461d591e1..5b5dc42006b 100644 --- a/plugins/logging/version +++ b/plugins/logging/version @@ -1 +1 @@ -1.5.25 +1.5.26 diff --git a/plugins/logging/writer.go b/plugins/logging/writer.go index 9b4dfbccdbb..2a706a9586c 100644 --- a/plugins/logging/writer.go +++ b/plugins/logging/writer.go @@ -162,8 +162,15 @@ func (p *LoggerPlugin) processBatch(batch []*writeQueueEntry) { // Individual fallback — isolate the bad entry instead of losing the whole batch for _, log := range logs { if err := p.store.BatchCreateIfNotExists(p.ctx, []*logstore.Log{log}); err != nil { - p.logger.Warn("individual insert failed for log %s: %v", log.ID, err) - p.droppedRequests.Add(1) + p.logger.Warn("individual insert failed for log %s, retrying without payload fields: %v", log.ID, err) + // Last resort: strip the parsed payload fields (one of them + // failed serialization) and keep the scalar row — a log + // without content beats a silently dropped request. + stripUnserializablePayloads(log) + if err := p.store.BatchCreateIfNotExists(p.ctx, []*logstore.Log{log}); err != nil { + p.logger.Warn("payload-stripped insert failed for log %s: %v", log.ID, err) + p.droppedRequests.Add(1) + } } } } @@ -280,6 +287,7 @@ func (p *LoggerPlugin) enqueueLogEntry(entry *logstore.Log, callback func(entry } defer func() { if r := recover(); r != nil { + p.logger.Error("recovered from a panic %v. dropping log request", r) // Channel was closed between the check and send; entry is dropped p.droppedRequests.Add(1) } diff --git a/plugins/maxim/changelog.md b/plugins/maxim/changelog.md index e69de29bb2d..a178a6f1d37 100644 --- a/plugins/maxim/changelog.md +++ b/plugins/maxim/changelog.md @@ -0,0 +1 @@ +- chore: upgraded core to v1.6.3 and framework to v1.4.3 diff --git a/plugins/maxim/version b/plugins/maxim/version index a92834b0ea9..757e60b6145 100644 --- a/plugins/maxim/version +++ b/plugins/maxim/version @@ -1 +1 @@ -1.6.25 +1.6.26 diff --git a/plugins/mocker/changelog.md b/plugins/mocker/changelog.md index e69de29bb2d..a178a6f1d37 100644 --- a/plugins/mocker/changelog.md +++ b/plugins/mocker/changelog.md @@ -0,0 +1 @@ +- chore: upgraded core to v1.6.3 and framework to v1.4.3 diff --git a/plugins/mocker/version b/plugins/mocker/version index 32461d591e1..5b5dc42006b 100644 --- a/plugins/mocker/version +++ b/plugins/mocker/version @@ -1 +1 @@ -1.5.25 +1.5.26 diff --git a/plugins/modelcatalogresolver/changelog.md b/plugins/modelcatalogresolver/changelog.md index e69de29bb2d..a178a6f1d37 100644 --- a/plugins/modelcatalogresolver/changelog.md +++ b/plugins/modelcatalogresolver/changelog.md @@ -0,0 +1 @@ +- chore: upgraded core to v1.6.3 and framework to v1.4.3 diff --git a/plugins/modelcatalogresolver/version b/plugins/modelcatalogresolver/version index af0b7ddbffd..238d6e882a0 100644 --- a/plugins/modelcatalogresolver/version +++ b/plugins/modelcatalogresolver/version @@ -1 +1 @@ -1.0.6 +1.0.7 diff --git a/plugins/otel/changelog.md b/plugins/otel/changelog.md index e69de29bb2d..a178a6f1d37 100644 --- a/plugins/otel/changelog.md +++ b/plugins/otel/changelog.md @@ -0,0 +1 @@ +- chore: upgraded core to v1.6.3 and framework to v1.4.3 diff --git a/plugins/otel/version b/plugins/otel/version index 1892b926767..31e5c843497 100644 --- a/plugins/otel/version +++ b/plugins/otel/version @@ -1 +1 @@ -1.3.2 +1.3.3 diff --git a/plugins/prompts/changelog.md b/plugins/prompts/changelog.md index e69de29bb2d..a178a6f1d37 100644 --- a/plugins/prompts/changelog.md +++ b/plugins/prompts/changelog.md @@ -0,0 +1 @@ +- chore: upgraded core to v1.6.3 and framework to v1.4.3 diff --git a/plugins/prompts/version b/plugins/prompts/version index 4a4127c371d..8955a0173eb 100644 --- a/plugins/prompts/version +++ b/plugins/prompts/version @@ -1 +1 @@ -1.0.25 +1.0.26 diff --git a/plugins/semanticcache/changelog.md b/plugins/semanticcache/changelog.md index e69de29bb2d..235ae03f057 100644 --- a/plugins/semanticcache/changelog.md +++ b/plugins/semanticcache/changelog.md @@ -0,0 +1,3 @@ +- fix: resolve internal embedding keys like external requests (#4903, closes #4756) (thanks [@nnNyx](https://github.com/nnNyx)!) +- fix: clear body-transport state for internal embedding requests via `ClearContextForInternalRequest` (#4918) +- chore: upgraded core to v1.6.3 and framework to v1.4.3 diff --git a/plugins/semanticcache/main.go b/plugins/semanticcache/main.go index 737ef7c96e9..5a9541cf111 100644 --- a/plugins/semanticcache/main.go +++ b/plugins/semanticcache/main.go @@ -603,6 +603,19 @@ func (plugin *Plugin) PostLLMHook(ctx *schemas.BifrostContext, res *schemas.Bifr embeddingToStore = nil } + // A store that requires vectors (Qdrant, Pinecone) rejects an empty-vector + // upsert ("Expected some vectors"). If embedding generation failed upstream + // (e.g. transient provider error, key resolution failure), skip the write + // rather than emit a broken point. cache_debug was already stamped above, + // so the miss stays observable. + if plugin.store.RequiresVectors() && len(embeddingToStore) == 0 { + // PostLLMHook runs once per streaming chunk; warn only once per request. + if !isStream || isFinalChunk { + plugin.logger.Warn("Skipping semantic cache write (namespace=%s, id=%s): store requires vectors but no embedding is available (embedding generation likely failed)", plugin.config.VectorStoreNamespace, storageID) + } + return res, nil, nil + } + plugin.writersWg.Add(1) go func() { defer plugin.writersWg.Done() diff --git a/plugins/semanticcache/plugin_paths_test.go b/plugins/semanticcache/plugin_paths_test.go index a1de790276e..404274aa2ca 100644 --- a/plugins/semanticcache/plugin_paths_test.go +++ b/plugins/semanticcache/plugin_paths_test.go @@ -562,3 +562,195 @@ func TestPreLLMHook_ConcurrentSameRequestID(t *testing.T) { t.Fatal("expected cache state to exist after concurrent PreLLMHook") } } + +// ----------------------------------------------------------------------------- +// Internal embedding request must not inherit the caller's key-routing state +// (issue #4756) or body-transport state. Otherwise governance/key-selection +// context resolved for the caller's chat provider leaks into the embedding +// request and rejects every embedding-provider key ("no keys found for +// provider"), and raw-body/large-payload passthrough state makes providers +// build the embedding request from the caller's body instead of embeddingReq. +// ----------------------------------------------------------------------------- + +// vectorRequiringStore is an observableStore that reports RequiresVectors()=true, +// mirroring dedicated vector DBs (Qdrant, Pinecone) that reject empty-vector +// upserts. All other behavior is inherited from observableStore. +type vectorRequiringStore struct { + *observableStore +} + +func (s *vectorRequiringStore) RequiresVectors() bool { return true } + +func TestGenerateEmbedding_ClearsInheritedKeyRoutingState(t *testing.T) { + plugin := newTestPlugin(t, newObservableStore()) + + var captured *schemas.BifrostContext + plugin.SetEmbeddingRequestExecutor(func(ctx *schemas.BifrostContext, _ *schemas.BifrostEmbeddingRequest) (*schemas.BifrostEmbeddingResponse, *schemas.BifrostError) { + captured = ctx + return &schemas.BifrostEmbeddingResponse{ + Data: []schemas.EmbeddingData{{ + Embedding: schemas.EmbeddingStruct{EmbeddingArray: []float64{0.1, 0.2, 0.3}}, + }}, + }, nil + }) + + // Simulate the caller's request context after governance resolved key + // routing for the CALLER's (chat) provider — none of these keys belong to + // the embedding provider. + ctx := scopedTestContext(t, "") + ctx.SetValue(schemas.BifrostContextKeyGovernanceIncludeOnlyKeys, []string{"chat-provider-key-id"}) + ctx.SetValue(schemas.BifrostContextKeyRoutingPinnedAPIKeyID, "chat-provider-key-id") + ctx.SetValue(schemas.BifrostContextKeyAPIKeyID, "chat-provider-key-id") + ctx.SetValue(schemas.BifrostContextKeyAPIKeyName, "chat-provider-key-name") + ctx.SetValue(schemas.BifrostContextKeyDirectKey, schemas.Key{ID: "direct-key-id"}) + ctx.SetValue(schemas.BifrostContextKeySkipKeySelection, true) + // Body-transport state from the caller's request. Inherited raw-body + // passthrough makes providers send the internal request's (absent) raw + // body instead of converting it; inherited large-payload mode streams the + // caller's original body to the embedding endpoint; inherited extra + // headers and URL path ride along to the embedding provider. + ctx.SetValue(schemas.BifrostContextKeyUseRawRequestBody, true) + ctx.SetValue(schemas.BifrostContextKeySendBackRawRequest, true) + ctx.SetValue(schemas.BifrostContextKeySendBackRawResponse, true) + ctx.SetValue(schemas.BifrostContextKeyPassthroughOverridesPresent, true) + ctx.SetValue(schemas.BifrostContextKeyLargePayloadMode, true) + ctx.SetValue(schemas.BifrostContextKeyLargeResponseMode, true) + ctx.SetValue(schemas.BifrostContextKeyExtraHeaders, map[string][]string{"x-custom": {"v"}}) + ctx.SetValue(schemas.BifrostContextKeyURLPath, "/v1/chat/completions") + + emb, _, err := plugin.generateEmbedding(ctx, "some text") + if err != nil { + t.Fatalf("generateEmbedding failed: %v", err) + } + if len(emb) == 0 { + t.Fatal("expected non-empty embedding") + } + if captured == nil { + t.Fatal("embedding executor was not invoked") + } + + // The internal embedding context must not carry the caller's key-routing + // or body-transport state, so key selection resolves against the embedding + // provider's own keys and providers build the request body from + // embeddingReq rather than the caller's raw/streamed body. + for _, key := range []schemas.BifrostContextKey{ + schemas.BifrostContextKeyGovernanceIncludeOnlyKeys, + schemas.BifrostContextKeyRoutingPinnedAPIKeyID, + schemas.BifrostContextKeyAPIKeyID, + schemas.BifrostContextKeyAPIKeyName, + schemas.BifrostContextKeyDirectKey, + schemas.BifrostContextKeySkipKeySelection, + schemas.BifrostContextKeyUseRawRequestBody, + schemas.BifrostContextKeySendBackRawRequest, + schemas.BifrostContextKeySendBackRawResponse, + schemas.BifrostContextKeyPassthroughOverridesPresent, + schemas.BifrostContextKeyLargePayloadMode, + schemas.BifrostContextKeyLargeResponseMode, + schemas.BifrostContextKeyExtraHeaders, + schemas.BifrostContextKeyURLPath, + } { + if v := captured.Value(key); v != nil { + t.Fatalf("expected %q cleared on internal embedding context, got %v", key, v) + } + } + + // SkipPluginPipeline must remain set — the internal embedding still bypasses + // the plugin pipeline; only the leaked key-routing state is cleared. + if skip, _ := captured.Value(schemas.BifrostContextKeySkipPluginPipeline).(bool); !skip { + t.Fatal("expected SkipPluginPipeline to remain set on internal embedding context") + } +} + +// ----------------------------------------------------------------------------- +// A failed embedding must not produce an empty-vector upsert on a store that +// requires vectors (issue #4756: Qdrant "Expected some vectors"). +// ----------------------------------------------------------------------------- + +func TestPostLLMHook_SkipsWriteWhenVectorRequiredButEmbeddingMissing(t *testing.T) { + base := newObservableStore() + plugin := newTestPlugin(t, &vectorRequiringStore{observableStore: base}) + // Embedding executor fails, mirroring the "no keys found" resolution error + // that leaves state.Embeddings unset. + plugin.SetEmbeddingRequestExecutor(func(_ *schemas.BifrostContext, _ *schemas.BifrostEmbeddingRequest) (*schemas.BifrostEmbeddingResponse, *schemas.BifrostError) { + return nil, &schemas.BifrostError{Error: &schemas.ErrorField{Message: "no keys found for provider"}} + }) + + ctx := CreateContextWithCacheKey(t, "") + req := &schemas.BifrostRequest{ + RequestType: schemas.ChatCompletionRequest, + ChatRequest: CreateBasicChatRequest("hello", 0.7, 50), + } + if _, sc, err := plugin.PreLLMHook(ctx, req); err != nil { + t.Fatalf("PreLLMHook failed: %v", err) + } else if sc != nil { + t.Fatalf("expected miss, got short-circuit %+v", sc) + } + + requestID, _ := ctx.Value(schemas.BifrostContextKeyRequestID).(string) + state := plugin.getCacheState(requestID) + if state == nil { + t.Fatal("expected cache state to exist") + } + if len(state.Embeddings) != 0 { + t.Fatalf("expected no embedding after failed generation, got %v", state.Embeddings) + } + + res := &schemas.BifrostResponse{ + ChatResponse: &schemas.BifrostChatResponse{ + ExtraFields: schemas.BifrostResponseExtraFields{RequestType: schemas.ChatCompletionRequest}, + }, + } + if _, _, err := plugin.PostLLMHook(ctx, res, nil); err != nil { + t.Fatalf("PostLLMHook failed: %v", err) + } + plugin.WaitForPendingOperations() + + base.mu.Lock() + defer base.mu.Unlock() + if len(base.addIDs) != 0 { + t.Fatalf("expected zero cache writes when embedding is missing on a vector-requiring store, got %d", len(base.addIDs)) + } +} + +func TestPostLLMHook_WritesWhenVectorRequiredAndEmbeddingPresent(t *testing.T) { + base := newObservableStore() + plugin := newTestPlugin(t, &vectorRequiringStore{observableStore: base}) + plugin.SetEmbeddingRequestExecutor(func(_ *schemas.BifrostContext, _ *schemas.BifrostEmbeddingRequest) (*schemas.BifrostEmbeddingResponse, *schemas.BifrostError) { + return &schemas.BifrostEmbeddingResponse{ + Data: []schemas.EmbeddingData{{ + Embedding: schemas.EmbeddingStruct{EmbeddingArray: []float64{0.1, 0.2, 0.3}}, + }}, + }, nil + }) + + ctx := CreateContextWithCacheKey(t, "") + req := &schemas.BifrostRequest{ + RequestType: schemas.ChatCompletionRequest, + ChatRequest: CreateBasicChatRequest("hello", 0.7, 50), + } + if _, _, err := plugin.PreLLMHook(ctx, req); err != nil { + t.Fatalf("PreLLMHook failed: %v", err) + } + + requestID, _ := ctx.Value(schemas.BifrostContextKeyRequestID).(string) + state := plugin.getCacheState(requestID) + if state == nil || len(state.Embeddings) == 0 { + t.Fatalf("expected embedding populated after successful generation, got %+v", state) + } + + res := &schemas.BifrostResponse{ + ChatResponse: &schemas.BifrostChatResponse{ + ExtraFields: schemas.BifrostResponseExtraFields{RequestType: schemas.ChatCompletionRequest}, + }, + } + if _, _, err := plugin.PostLLMHook(ctx, res, nil); err != nil { + t.Fatalf("PostLLMHook failed: %v", err) + } + plugin.WaitForPendingOperations() + + base.mu.Lock() + defer base.mu.Unlock() + if len(base.addIDs) != 1 { + t.Fatalf("expected one cache write when embedding is present, got %d", len(base.addIDs)) + } +} diff --git a/plugins/semanticcache/search.go b/plugins/semanticcache/search.go index 79b4c1b32ad..4e10e73b9b9 100644 --- a/plugins/semanticcache/search.go +++ b/plugins/semanticcache/search.go @@ -146,6 +146,12 @@ func (plugin *Plugin) generateEmbedding(ctx *schemas.BifrostContext, text string // released back to its sync.Pool — see core/schemas.ReleasePluginScope. defer embeddingCtx.Cancel() embeddingCtx.SetValue(schemas.BifrostContextKeySkipPluginPipeline, true) + // The embedding request targets the plugin's own configured embedding + // provider/model, not the caller's — and because it skips the plugin + // pipeline, routing state is never re-resolved for it. Shed the caller's + // key-routing and body-transport state so the request behaves like a + // fresh external /v1/embeddings call. + bifrost.ClearContextForInternalRequest(embeddingCtx) if plugin.embeddingRequestExecutor == nil { return nil, 0, fmt.Errorf("embedding request executor is not configured") } diff --git a/plugins/semanticcache/version b/plugins/semanticcache/version index 32461d591e1..5b5dc42006b 100644 --- a/plugins/semanticcache/version +++ b/plugins/semanticcache/version @@ -1 +1 @@ -1.5.25 +1.5.26 diff --git a/plugins/telemetry/changelog.md b/plugins/telemetry/changelog.md index e69de29bb2d..a178a6f1d37 100644 --- a/plugins/telemetry/changelog.md +++ b/plugins/telemetry/changelog.md @@ -0,0 +1 @@ +- chore: upgraded core to v1.6.3 and framework to v1.4.3 diff --git a/plugins/telemetry/version b/plugins/telemetry/version index 32461d591e1..5b5dc42006b 100644 --- a/plugins/telemetry/version +++ b/plugins/telemetry/version @@ -1 +1 @@ -1.5.25 +1.5.26 diff --git a/scripts/bifrost-migration-cli/.gitignore b/scripts/bifrost-migration-cli/.gitignore deleted file mode 100644 index 9f10d22fcc2..00000000000 --- a/scripts/bifrost-migration-cli/.gitignore +++ /dev/null @@ -1 +0,0 @@ -bifrost-migration-cli diff --git a/scripts/bifrost-migration-cli/model.go b/scripts/bifrost-migration-cli/model.go index 0abcd66ccc8..95486d82064 100644 --- a/scripts/bifrost-migration-cli/model.go +++ b/scripts/bifrost-migration-cli/model.go @@ -121,6 +121,7 @@ var standardProviders = map[string]bool{ "bedrock": true, "cerebras": true, "cohere": true, + "deepseek": true, "elevenlabs": true, "fireworks": true, "gemini": true, diff --git a/tests/cmd/e2eseed/go.mod b/tests/cmd/e2eseed/go.mod index 4f5a0c3b461..42fb6b5114d 100644 --- a/tests/cmd/e2eseed/go.mod +++ b/tests/cmd/e2eseed/go.mod @@ -23,6 +23,8 @@ require ( github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.13.1 // indirect github.com/Azure/azure-sdk-for-go/sdk/internal v1.11.2 // indirect github.com/AzureAD/microsoft-authentication-library-for-go v1.6.0 // indirect + github.com/ClickHouse/ch-go v0.61.5 // indirect + github.com/ClickHouse/clickhouse-go/v2 v2.30.0 // indirect github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.31.0 // indirect github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.55.0 // indirect github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/resourcemapping v0.55.0 // indirect @@ -57,6 +59,8 @@ require ( github.com/envoyproxy/go-control-plane/envoy v1.37.0 // indirect github.com/envoyproxy/protoc-gen-validate v1.3.3 // indirect github.com/felixge/httpsnoop v1.0.4 // indirect + github.com/go-faster/city v1.0.1 // indirect + github.com/go-faster/errors v0.7.1 // indirect github.com/go-jose/go-jose/v4 v4.1.4 // indirect github.com/go-logr/logr v1.4.3 // indirect github.com/go-logr/stdr v1.2.2 // indirect @@ -65,6 +69,7 @@ require ( github.com/google/uuid v1.6.0 // indirect github.com/googleapis/enterprise-certificate-proxy v0.3.16 // indirect github.com/googleapis/gax-go/v2 v2.22.0 // indirect + github.com/hashicorp/go-version v1.6.0 // indirect github.com/invopop/jsonschema v0.13.0 // indirect github.com/jackc/pgpassfile v1.0.0 // indirect github.com/jackc/pgservicefile v0.0.0-20240606120523-5a60cdf6a761 // indirect @@ -80,11 +85,16 @@ require ( github.com/mattn/go-colorable v0.1.14 // indirect github.com/mattn/go-isatty v0.0.20 // indirect github.com/mattn/go-sqlite3 v1.14.32 // indirect - github.com/maximhq/bifrost/core v1.6.1 // indirect + github.com/maximhq/bifrost/core v1.6.2 // indirect github.com/maximhq/bifrost/framework v1.3.16 // indirect + github.com/paulmach/orb v0.11.1 // indirect + github.com/pierrec/lz4/v4 v4.1.21 // indirect github.com/pkg/browser v0.0.0-20240102092130-5ac0b6a4141c // indirect + github.com/pkg/errors v0.9.1 // indirect github.com/planetscale/vtprotobuf v0.6.1-0.20240319094008-0393e58bdf10 // indirect github.com/rs/zerolog v1.34.0 // indirect + github.com/segmentio/asm v1.2.0 // indirect + github.com/shopspring/decimal v1.4.0 // indirect github.com/spf13/cast v1.10.0 // indirect github.com/spiffe/go-spiffe/v2 v2.6.0 // indirect github.com/tidwall/gjson v1.18.0 // indirect @@ -121,6 +131,7 @@ require ( google.golang.org/grpc v1.81.1 // indirect google.golang.org/protobuf v1.36.12-0.20260120151049-f2248ac996af // indirect gopkg.in/yaml.v3 v3.0.1 // indirect + gorm.io/driver/clickhouse v0.7.0 // indirect gorm.io/driver/postgres v1.6.0 // indirect gorm.io/driver/sqlite v1.6.0 // indirect gorm.io/gorm v1.31.1 // indirect diff --git a/tests/cmd/e2eseed/go.sum b/tests/cmd/e2eseed/go.sum index 4030f84974b..1c0a3064924 100644 --- a/tests/cmd/e2eseed/go.sum +++ b/tests/cmd/e2eseed/go.sum @@ -32,6 +32,8 @@ github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1 h1:WJ github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1/go.mod h1:tCcJZ0uHAmvjsVYzEFivsRTN00oz5BEsRgQHu5JZ9WE= github.com/AzureAD/microsoft-authentication-library-for-go v1.6.0 h1:XRzhVemXdgvJqCH0sFfrBUTnUJSBrBf7++ypk+twtRs= github.com/AzureAD/microsoft-authentication-library-for-go v1.6.0/go.mod h1:HKpQxkWaGLJ+D/5H8QRpyQXA1eKjxkFlOMwck5+33Jk= +github.com/ClickHouse/ch-go v0.61.5 h1:zwR8QbYI0tsMiEcze/uIMK+Tz1D3XZXLdNrlaOpeEI4= +github.com/ClickHouse/clickhouse-go/v2 v2.30.0 h1:AG4D/hW39qa58+JHQIFOSnxyL46H6h2lrmGGk17dhFo= github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.31.0 h1:DHa2U07rk8syqvCge0QIGMCE1WxGj9njT44GH7zNJLQ= github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.31.0/go.mod h1:P4WPRUkOhJC13W//jWpyfJNDAIpvRbAUIYLX/4jtlE0= github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.55.0 h1:UnDZ/zFfG1JhH/DqxIZYU/1CUAlTUScoXD/LcM2Ykk8= @@ -115,6 +117,8 @@ github.com/felixge/httpsnoop v1.0.4 h1:NFTV2Zj1bL4mc9sqWACXbQFVBBg2W3GPvqp8/ESS2 github.com/felixge/httpsnoop v1.0.4/go.mod h1:m8KPJKqk1gH5J9DgRY2ASl2lWCfGKXixSwevea8zH2U= github.com/frankban/quicktest v1.14.6 h1:7Xjx+VpznH+oBnejlPUj8oUpdxnVs4f8XU8WnHkI4W8= github.com/frankban/quicktest v1.14.6/go.mod h1:4ptaffx2x8+WTWXmUCuVU6aPUX1/Mz7zb5vbUoiM6w0= +github.com/go-faster/city v1.0.1 h1:4WAxSZ3V2Ws4QRDrscLEDcibJY8uf41H6AhXDrNDcGw= +github.com/go-faster/errors v0.7.1 h1:MkJTnDoEdi9pDabt1dpWf7AA8/BaSYZqibYyhZ20AYg= github.com/go-jose/go-jose/v4 v4.1.4 h1:moDMcTHmvE6Groj34emNPLs/qtYXRVcd6S7NHbHz3kA= github.com/go-jose/go-jose/v4 v4.1.4/go.mod h1:x4oUasVrzR7071A4TnHLGSPpNOm2a21K9Kf04k1rs08= github.com/go-logr/logr v1.2.2/go.mod h1:jdQByPbusPIv2/zmleS9BjJVeZ6kBagPoEUsqbVz/1A= @@ -141,6 +145,7 @@ github.com/googleapis/gax-go/v2 v2.22.0 h1:PjIWBpgGIVKGoCXuiCoP64altEJCj3/Ei+kSU github.com/googleapis/gax-go/v2 v2.22.0/go.mod h1:irWBbALSr0Sk3qlqb9SyJ1h68WjgeFuiOzI4Rqw5+aY= github.com/hajimehoshi/go-mp3 v0.3.4 h1:NUP7pBYH8OguP4diaTZ9wJbUbk3tC0KlfzsEpWmYj68= github.com/hajimehoshi/go-mp3 v0.3.4/go.mod h1:fRtZraRFcWb0pu7ok0LqyFhCUrPeMsGRSVop0eemFmo= +github.com/hashicorp/go-version v1.6.0 h1:feTTfFNnjP967rlCxM/I9g701jU+RN74YKx2mOkIeek= github.com/invopop/jsonschema v0.13.0 h1:KvpoAJWEjR3uD9Kbm2HWJmqsEaHt8lBUpd0qHcIi21E= github.com/invopop/jsonschema v0.13.0/go.mod h1:ffZ5Km5SWWRAIN6wbDXItl95euhFz2uON45H2qjYt+0= github.com/jackc/pgpassfile v1.0.0 h1:/6Hmqy13Ss2zCq62VdNG8tM1wchn8zjSGOBJ6icpsIM= @@ -180,8 +185,11 @@ github.com/mattn/go-isatty v0.0.20 h1:xfD0iDuEKnDkl03q4limB+vH+GxLEtL/jb4xVJSWWE github.com/mattn/go-isatty v0.0.20/go.mod h1:W+V8PltTTMOvKvAeJH7IuucS94S2C6jfK/D7dTCTo3Y= github.com/mattn/go-sqlite3 v1.14.32 h1:JD12Ag3oLy1zQA+BNn74xRgaBbdhbNIDYvQUEuuErjs= github.com/mattn/go-sqlite3 v1.14.32/go.mod h1:Uh1q+B4BYcTPb+yiD3kU8Ct7aC0hY9fxUwlHK0RXw+Y= +github.com/paulmach/orb v0.11.1 h1:3koVegMC4X/WeiXYz9iswopaTwMem53NzTJuTF20JzU= +github.com/pierrec/lz4/v4 v4.1.21 h1:yOVMLb6qSIDP67pl/5F7RepeKYu/VmTyEXvuMI5d9mQ= github.com/pkg/browser v0.0.0-20240102092130-5ac0b6a4141c h1:+mdjkGKdHQG3305AYmdv1U2eRNDiU2ErMBj1gwrq8eQ= github.com/pkg/browser v0.0.0-20240102092130-5ac0b6a4141c/go.mod h1:7rwL4CYBLnjLxUqIJNnCWiEdr3bn6IUYi15bNlnbCCU= +github.com/pkg/errors v0.9.1 h1:FEBLx1zS214owpjy7qsBeixbURkuhQAwrK5UwLGTwt4= github.com/pkg/errors v0.9.1/go.mod h1:bwawxfHBFNV+L2hUp1rHADufV3IMtnDRdf1r5NINEl0= github.com/planetscale/vtprotobuf v0.6.1-0.20240319094008-0393e58bdf10 h1:GFCKgmp0tecUJ0sJuv4pzYCqS9+RGSn52M3FUwPs+uo= github.com/planetscale/vtprotobuf v0.6.1-0.20240319094008-0393e58bdf10/go.mod h1:t/avpk3KcrXxUnYOhZhMXJlSEyie6gQbtLq5NM3loB8= @@ -195,6 +203,8 @@ github.com/rs/zerolog v1.34.0 h1:k43nTLIwcTVQAncfCw4KZ2VY6ukYoZaBPNOE8txlOeY= github.com/rs/zerolog v1.34.0/go.mod h1:bJsvje4Z08ROH4Nhs5iH600c3IkWhwp44iRc54W6wYQ= github.com/savsgio/gotils v0.0.0-20250408102913-196191ec6287 h1:qIQ0tWF9vxGtkJa24bR+2i53WBCz1nW/Pc47oVYauC4= github.com/savsgio/gotils v0.0.0-20250408102913-196191ec6287/go.mod h1:sM7Mt7uEoCeFSCBM+qBrqvEo+/9vdmj19wzp3yzUhmg= +github.com/segmentio/asm v1.2.0 h1:9BQrFxC+YOHJlTlHGkTrFWf59nbL3XnCoFLTwDCI7ys= +github.com/shopspring/decimal v1.4.0 h1:bxl37RwXBklmTi0C79JfXCEBD1cqqHt0bbgBAGFp81k= github.com/spf13/cast v1.10.0 h1:h2x0u2shc1QuLHfxi+cTJvs30+ZAHOGRic8uyGTDWxY= github.com/spf13/cast v1.10.0/go.mod h1:jNfB8QC9IA6ZuY2ZjDp0KtFO2LZZlg4S/7bzP6qqeHo= github.com/spiffe/go-spiffe/v2 v2.6.0 h1:l+DolpxNWYgruGQVV0xsfeya3CsC7m8iBzDnMpsbLuo= @@ -294,6 +304,7 @@ gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EV gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= +gorm.io/driver/clickhouse v0.7.0 h1:BCrqvgONayvZRgtuA6hdya+eAW5P2QVagV3OlEp1vtA= gorm.io/driver/postgres v1.6.0 h1:2dxzU8xJ+ivvqTRph34QX+WrRaJlmfyPqXmoGVjMBa4= gorm.io/driver/postgres v1.6.0/go.mod h1:vUw0mrGgrTK+uPHEhAdV4sfFELrByKVGnaVRkXDhtWo= gorm.io/driver/sqlite v1.6.0 h1:WHRRrIiulaPiPFmDcod6prc4l2VGVWHz80KspNsxSfQ= diff --git a/tests/cmd/seed/go.mod b/tests/cmd/seed/go.mod index 483fde26b42..e2ba4983a1a 100644 --- a/tests/cmd/seed/go.mod +++ b/tests/cmd/seed/go.mod @@ -8,7 +8,7 @@ replace ( ) require ( - github.com/maximhq/bifrost/core v1.6.1 + github.com/maximhq/bifrost/core v1.6.2 github.com/maximhq/bifrost/framework v1.3.16 gorm.io/driver/postgres v1.6.0 gorm.io/driver/sqlite v1.6.0 @@ -28,6 +28,8 @@ require ( github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.13.1 // indirect github.com/Azure/azure-sdk-for-go/sdk/internal v1.11.2 // indirect github.com/AzureAD/microsoft-authentication-library-for-go v1.6.0 // indirect + github.com/ClickHouse/ch-go v0.61.5 // indirect + github.com/ClickHouse/clickhouse-go/v2 v2.30.0 // indirect github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.31.0 // indirect github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.55.0 // indirect github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/resourcemapping v0.55.0 // indirect @@ -62,6 +64,8 @@ require ( github.com/envoyproxy/go-control-plane/envoy v1.37.0 // indirect github.com/envoyproxy/protoc-gen-validate v1.3.3 // indirect github.com/felixge/httpsnoop v1.0.4 // indirect + github.com/go-faster/city v1.0.1 // indirect + github.com/go-faster/errors v0.7.1 // indirect github.com/go-jose/go-jose/v4 v4.1.4 // indirect github.com/go-logr/logr v1.4.3 // indirect github.com/go-logr/stdr v1.2.2 // indirect @@ -70,6 +74,7 @@ require ( github.com/google/uuid v1.6.0 // indirect github.com/googleapis/enterprise-certificate-proxy v0.3.16 // indirect github.com/googleapis/gax-go/v2 v2.22.0 // indirect + github.com/hashicorp/go-version v1.6.0 // indirect github.com/invopop/jsonschema v0.13.0 // indirect github.com/jackc/pgpassfile v1.0.0 // indirect github.com/jackc/pgservicefile v0.0.0-20240606120523-5a60cdf6a761 // indirect @@ -85,9 +90,14 @@ require ( github.com/mattn/go-colorable v0.1.14 // indirect github.com/mattn/go-isatty v0.0.20 // indirect github.com/mattn/go-sqlite3 v1.14.32 // indirect + github.com/paulmach/orb v0.11.1 // indirect + github.com/pierrec/lz4/v4 v4.1.21 // indirect github.com/pkg/browser v0.0.0-20240102092130-5ac0b6a4141c // indirect + github.com/pkg/errors v0.9.1 // indirect github.com/planetscale/vtprotobuf v0.6.1-0.20240319094008-0393e58bdf10 // indirect github.com/rs/zerolog v1.34.0 // indirect + github.com/segmentio/asm v1.2.0 // indirect + github.com/shopspring/decimal v1.4.0 // indirect github.com/spf13/cast v1.10.0 // indirect github.com/spiffe/go-spiffe/v2 v2.6.0 // indirect github.com/tidwall/gjson v1.18.0 // indirect @@ -124,4 +134,5 @@ require ( google.golang.org/grpc v1.81.1 // indirect google.golang.org/protobuf v1.36.12-0.20260120151049-f2248ac996af // indirect gopkg.in/yaml.v3 v3.0.1 // indirect + gorm.io/driver/clickhouse v0.7.0 // indirect ) diff --git a/tests/cmd/seed/go.sum b/tests/cmd/seed/go.sum index 4030f84974b..1c0a3064924 100644 --- a/tests/cmd/seed/go.sum +++ b/tests/cmd/seed/go.sum @@ -32,6 +32,8 @@ github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1 h1:WJ github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1/go.mod h1:tCcJZ0uHAmvjsVYzEFivsRTN00oz5BEsRgQHu5JZ9WE= github.com/AzureAD/microsoft-authentication-library-for-go v1.6.0 h1:XRzhVemXdgvJqCH0sFfrBUTnUJSBrBf7++ypk+twtRs= github.com/AzureAD/microsoft-authentication-library-for-go v1.6.0/go.mod h1:HKpQxkWaGLJ+D/5H8QRpyQXA1eKjxkFlOMwck5+33Jk= +github.com/ClickHouse/ch-go v0.61.5 h1:zwR8QbYI0tsMiEcze/uIMK+Tz1D3XZXLdNrlaOpeEI4= +github.com/ClickHouse/clickhouse-go/v2 v2.30.0 h1:AG4D/hW39qa58+JHQIFOSnxyL46H6h2lrmGGk17dhFo= github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.31.0 h1:DHa2U07rk8syqvCge0QIGMCE1WxGj9njT44GH7zNJLQ= github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.31.0/go.mod h1:P4WPRUkOhJC13W//jWpyfJNDAIpvRbAUIYLX/4jtlE0= github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.55.0 h1:UnDZ/zFfG1JhH/DqxIZYU/1CUAlTUScoXD/LcM2Ykk8= @@ -115,6 +117,8 @@ github.com/felixge/httpsnoop v1.0.4 h1:NFTV2Zj1bL4mc9sqWACXbQFVBBg2W3GPvqp8/ESS2 github.com/felixge/httpsnoop v1.0.4/go.mod h1:m8KPJKqk1gH5J9DgRY2ASl2lWCfGKXixSwevea8zH2U= github.com/frankban/quicktest v1.14.6 h1:7Xjx+VpznH+oBnejlPUj8oUpdxnVs4f8XU8WnHkI4W8= github.com/frankban/quicktest v1.14.6/go.mod h1:4ptaffx2x8+WTWXmUCuVU6aPUX1/Mz7zb5vbUoiM6w0= +github.com/go-faster/city v1.0.1 h1:4WAxSZ3V2Ws4QRDrscLEDcibJY8uf41H6AhXDrNDcGw= +github.com/go-faster/errors v0.7.1 h1:MkJTnDoEdi9pDabt1dpWf7AA8/BaSYZqibYyhZ20AYg= github.com/go-jose/go-jose/v4 v4.1.4 h1:moDMcTHmvE6Groj34emNPLs/qtYXRVcd6S7NHbHz3kA= github.com/go-jose/go-jose/v4 v4.1.4/go.mod h1:x4oUasVrzR7071A4TnHLGSPpNOm2a21K9Kf04k1rs08= github.com/go-logr/logr v1.2.2/go.mod h1:jdQByPbusPIv2/zmleS9BjJVeZ6kBagPoEUsqbVz/1A= @@ -141,6 +145,7 @@ github.com/googleapis/gax-go/v2 v2.22.0 h1:PjIWBpgGIVKGoCXuiCoP64altEJCj3/Ei+kSU github.com/googleapis/gax-go/v2 v2.22.0/go.mod h1:irWBbALSr0Sk3qlqb9SyJ1h68WjgeFuiOzI4Rqw5+aY= github.com/hajimehoshi/go-mp3 v0.3.4 h1:NUP7pBYH8OguP4diaTZ9wJbUbk3tC0KlfzsEpWmYj68= github.com/hajimehoshi/go-mp3 v0.3.4/go.mod h1:fRtZraRFcWb0pu7ok0LqyFhCUrPeMsGRSVop0eemFmo= +github.com/hashicorp/go-version v1.6.0 h1:feTTfFNnjP967rlCxM/I9g701jU+RN74YKx2mOkIeek= github.com/invopop/jsonschema v0.13.0 h1:KvpoAJWEjR3uD9Kbm2HWJmqsEaHt8lBUpd0qHcIi21E= github.com/invopop/jsonschema v0.13.0/go.mod h1:ffZ5Km5SWWRAIN6wbDXItl95euhFz2uON45H2qjYt+0= github.com/jackc/pgpassfile v1.0.0 h1:/6Hmqy13Ss2zCq62VdNG8tM1wchn8zjSGOBJ6icpsIM= @@ -180,8 +185,11 @@ github.com/mattn/go-isatty v0.0.20 h1:xfD0iDuEKnDkl03q4limB+vH+GxLEtL/jb4xVJSWWE github.com/mattn/go-isatty v0.0.20/go.mod h1:W+V8PltTTMOvKvAeJH7IuucS94S2C6jfK/D7dTCTo3Y= github.com/mattn/go-sqlite3 v1.14.32 h1:JD12Ag3oLy1zQA+BNn74xRgaBbdhbNIDYvQUEuuErjs= github.com/mattn/go-sqlite3 v1.14.32/go.mod h1:Uh1q+B4BYcTPb+yiD3kU8Ct7aC0hY9fxUwlHK0RXw+Y= +github.com/paulmach/orb v0.11.1 h1:3koVegMC4X/WeiXYz9iswopaTwMem53NzTJuTF20JzU= +github.com/pierrec/lz4/v4 v4.1.21 h1:yOVMLb6qSIDP67pl/5F7RepeKYu/VmTyEXvuMI5d9mQ= github.com/pkg/browser v0.0.0-20240102092130-5ac0b6a4141c h1:+mdjkGKdHQG3305AYmdv1U2eRNDiU2ErMBj1gwrq8eQ= github.com/pkg/browser v0.0.0-20240102092130-5ac0b6a4141c/go.mod h1:7rwL4CYBLnjLxUqIJNnCWiEdr3bn6IUYi15bNlnbCCU= +github.com/pkg/errors v0.9.1 h1:FEBLx1zS214owpjy7qsBeixbURkuhQAwrK5UwLGTwt4= github.com/pkg/errors v0.9.1/go.mod h1:bwawxfHBFNV+L2hUp1rHADufV3IMtnDRdf1r5NINEl0= github.com/planetscale/vtprotobuf v0.6.1-0.20240319094008-0393e58bdf10 h1:GFCKgmp0tecUJ0sJuv4pzYCqS9+RGSn52M3FUwPs+uo= github.com/planetscale/vtprotobuf v0.6.1-0.20240319094008-0393e58bdf10/go.mod h1:t/avpk3KcrXxUnYOhZhMXJlSEyie6gQbtLq5NM3loB8= @@ -195,6 +203,8 @@ github.com/rs/zerolog v1.34.0 h1:k43nTLIwcTVQAncfCw4KZ2VY6ukYoZaBPNOE8txlOeY= github.com/rs/zerolog v1.34.0/go.mod h1:bJsvje4Z08ROH4Nhs5iH600c3IkWhwp44iRc54W6wYQ= github.com/savsgio/gotils v0.0.0-20250408102913-196191ec6287 h1:qIQ0tWF9vxGtkJa24bR+2i53WBCz1nW/Pc47oVYauC4= github.com/savsgio/gotils v0.0.0-20250408102913-196191ec6287/go.mod h1:sM7Mt7uEoCeFSCBM+qBrqvEo+/9vdmj19wzp3yzUhmg= +github.com/segmentio/asm v1.2.0 h1:9BQrFxC+YOHJlTlHGkTrFWf59nbL3XnCoFLTwDCI7ys= +github.com/shopspring/decimal v1.4.0 h1:bxl37RwXBklmTi0C79JfXCEBD1cqqHt0bbgBAGFp81k= github.com/spf13/cast v1.10.0 h1:h2x0u2shc1QuLHfxi+cTJvs30+ZAHOGRic8uyGTDWxY= github.com/spf13/cast v1.10.0/go.mod h1:jNfB8QC9IA6ZuY2ZjDp0KtFO2LZZlg4S/7bzP6qqeHo= github.com/spiffe/go-spiffe/v2 v2.6.0 h1:l+DolpxNWYgruGQVV0xsfeya3CsC7m8iBzDnMpsbLuo= @@ -294,6 +304,7 @@ gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EV gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= +gorm.io/driver/clickhouse v0.7.0 h1:BCrqvgONayvZRgtuA6hdya+eAW5P2QVagV3OlEp1vtA= gorm.io/driver/postgres v1.6.0 h1:2dxzU8xJ+ivvqTRph34QX+WrRaJlmfyPqXmoGVjMBa4= gorm.io/driver/postgres v1.6.0/go.mod h1:vUw0mrGgrTK+uPHEhAdV4sfFELrByKVGnaVRkXDhtWo= gorm.io/driver/sqlite v1.6.0 h1:WHRRrIiulaPiPFmDcod6prc4l2VGVWHz80KspNsxSfQ= diff --git a/tests/cmd/seed/seed.go b/tests/cmd/seed/seed.go index f60c2a9cc55..1ba84ec86a9 100644 --- a/tests/cmd/seed/seed.go +++ b/tests/cmd/seed/seed.go @@ -613,9 +613,9 @@ func seedGovernance(ctx context.Context, db *gorm.DB, prefix string, now time.Ti } } for _, vk := range []tables.TableVirtualKey{ - {ID: prefix + "-vk-user-team", Name: "E2E User Team VK", Value: prefix + "-vk-user-team-secret", IsActive: &active, TeamID: &tiggingsTeam, CreatedAt: now, UpdatedAt: now}, - {ID: prefix + "-vk-team-only", Name: "E2E Team Only VK", Value: prefix + "-vk-team-only-secret", IsActive: &active, TeamID: &tiggingsTeam, CreatedAt: now, UpdatedAt: now}, - {ID: prefix + "-vk-outside", Name: "E2E Outside VK", Value: prefix + "-vk-outside-secret", IsActive: &active, TeamID: &outsideTeam, CreatedAt: now, UpdatedAt: now}, + {ID: prefix + "-vk-user-team", Name: "E2E User Team VK", Value: *schemas.NewSecretVar(prefix + "-vk-user-team-secret"), IsActive: &active, TeamID: &tiggingsTeam, CreatedAt: now, UpdatedAt: now}, + {ID: prefix + "-vk-team-only", Name: "E2E Team Only VK", Value: *schemas.NewSecretVar(prefix + "-vk-team-only-secret"), IsActive: &active, TeamID: &tiggingsTeam, CreatedAt: now, UpdatedAt: now}, + {ID: prefix + "-vk-outside", Name: "E2E Outside VK", Value: *schemas.NewSecretVar(prefix + "-vk-outside-secret"), IsActive: &active, TeamID: &outsideTeam, CreatedAt: now, UpdatedAt: now}, } { if err := db.WithContext(ctx).Where("id = ?", vk.ID).Assign(vk).FirstOrCreate(&vk).Error; err != nil { return err diff --git a/tests/cmd/seedvks/go.mod b/tests/cmd/seedvks/go.mod index 33b8ed074c6..ee4f835d309 100644 --- a/tests/cmd/seedvks/go.mod +++ b/tests/cmd/seedvks/go.mod @@ -9,7 +9,7 @@ replace ( require ( github.com/google/uuid v1.6.0 - github.com/maximhq/bifrost/core v1.6.1 + github.com/maximhq/bifrost/core v1.6.2 github.com/maximhq/bifrost/framework v1.3.16 gorm.io/driver/postgres v1.6.0 gorm.io/gorm v1.31.1 diff --git a/tests/cmd/seedvks/main.go b/tests/cmd/seedvks/main.go index 64d8feb429e..701239ed172 100644 --- a/tests/cmd/seedvks/main.go +++ b/tests/cmd/seedvks/main.go @@ -135,7 +135,7 @@ func generateBatch(prefix string, start, n int, now time.Time) []configstoreTabl rows = append(rows, configstoreTables.TableVirtualKey{ ID: uuid.NewString(), Name: fmt.Sprintf("%s-%d", prefix, idx), - Value: virtualKeyPrefix + uuid.NewString(), + Value: *schemas.NewSecretVar(virtualKeyPrefix + uuid.NewString()), CreatedAt: now, UpdatedAt: now, }) diff --git a/tests/e2e/api/README.md b/tests/e2e/api/README.md index 21c773d5129..dc1f0c7e989 100644 --- a/tests/e2e/api/README.md +++ b/tests/e2e/api/README.md @@ -228,6 +228,46 @@ Run locally (from this directory): # options: --port (default 8090), --html, --json, --verbose, --bail ``` +### MCP Auth Tests + +| Path | Description | +|------|-------------| +| `collections/bifrost-v1-mcp-auth.postman_collection.json` | Asserts inbound `/mcp` authentication across the three server auth modes (`headers` / `both` / `oauth`): discovery gating, the credential connect matrix, the full issuance flow, refresh rotation + family revocation, the revocation window, and the runtime `headers`→`both` upgrade. | +| `runners/individual/run-newman-mcp-auth-tests.sh` | Builds + starts the upstream MCP server, then boots a fresh server per `client.mcp_server_auth_mode` and runs the collection against each. | + +Like the auth-matrix runner, this one **boots its own servers** — each mode needs a +different boot config. It also builds and starts the upstream MCP server +(`examples/mcps/http-no-ping-server`) so `/mcp` exposes real tools, and pre-seeds it +as an MCP client plus two virtual keys (one active, one inactive). It requires a +built `bifrost-http` binary. + +The collection's test scripts branch on the `auth_mode` env-var, so a single +collection encodes the full matrix. Per mode it asserts: + +- **`headers` (default):** discovery endpoints 404; every virtual-key credential + (`x-bf-vk`, `Authorization: Bearer `, `x-api-key`) connects exactly as before; + anonymous connects when auth is not enforced; an inactive key never connects. The + OAuth surface is invisible. +- **`both`:** every header-credential outcome is identical to `headers`, and only + *adds* JWT acceptance + live discovery; an invalid JWT is rejected with + `WWW-Authenticate`. +- **`oauth`:** header credentials and anonymous are rejected (401 + + `WWW-Authenticate`); only issued JWTs connect. + +In `both`/`oauth` it also runs the end-to-end issuance flow (dynamic client +registration, PKCE-S256 authorize, consent bound to a virtual key or to a +server-minted session, token exchange, JWT connect), then refresh rotation with +stolen-token family revocation, the revocation window (a revoked grant stops refresh +while its already-issued access token keeps working until expiry), and — from the +`headers` boot — the runtime `headers`→`both` upgrade. + +Run locally (from this directory): + +```bash +./runners/individual/run-newman-mcp-auth-tests.sh --binary /path/to/bifrost-http +# options: --port (default 8090), --mcp-port (default 3001), --html, --json, --verbose, --bail +``` + ### Test Success Criteria A request **passes** if either: diff --git a/tests/e2e/api/collections/bifrost-v1-mcp-auth.postman_collection.json b/tests/e2e/api/collections/bifrost-v1-mcp-auth.postman_collection.json new file mode 100644 index 00000000000..c958e84d784 --- /dev/null +++ b/tests/e2e/api/collections/bifrost-v1-mcp-auth.postman_collection.json @@ -0,0 +1,1329 @@ +{ + "info": { + "name": "Bifrost V1 - MCP Auth", + "description": "Validates inbound /mcp authentication across the three server auth modes (headers | both | oauth), driven by the auth_mode env-var set per boot config. The runner boots a fresh server per mode with two pre-seeded virtual keys (one active, one inactive) and an upstream MCP client. Core guarantees: (1) in headers mode every virtual-key credential connects exactly as before and the OAuth discovery surface is invisible (404); (2) both mode keeps every header-credential outcome identical and only ADDS JWT acceptance + discovery; (3) oauth mode rejects header credentials outright; (4) an inactive virtual key never connects, in any mode; (5) discovery documents are spec-shaped when served.", + "schema": "https://schema.getpostman.com/json/collection/v2.1.0/collection.json" + }, + "variable": [ + { "key": "base_url", "value": "http://localhost:8090", "type": "string" }, + { "key": "auth_mode", "value": "headers", "type": "string" }, + { "key": "vk_value", "value": "sk-bf-mcp-test-key", "type": "string" }, + { "key": "vk_inactive_value", "value": "sk-bf-mcp-inactive-key", "type": "string" }, + { "key": "mcp_issuer", "value": "http://localhost:8090", "type": "string" } + ], + "item": [ + { + "name": "Discovery gating", + "item": [ + { + "name": "Protected resource metadata", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "var code = pm.response.code;", + "if (mode === 'headers') {", + " pm.test('PRM is hidden in headers mode (404)', function () {", + " pm.expect(code, pm.response.text()).to.equal(404);", + " });", + " return;", + "}", + "pm.test('PRM is served (200)', function () { pm.expect(code).to.equal(200); });", + "var b = pm.response.json();", + "var issuer = pm.variables.get('mcp_issuer');", + "pm.test('resource is the /mcp URL (RFC 9728)', function () {", + " pm.expect(b.resource).to.equal(issuer + '/mcp');", + "});", + "pm.test('authorization_servers lists this issuer', function () {", + " pm.expect(b.authorization_servers).to.include(issuer);", + "});" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/.well-known/oauth-protected-resource", + "host": ["{{base_url}}"], + "path": [".well-known", "oauth-protected-resource"] + } + } + }, + { + "name": "Authorization server metadata", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "var code = pm.response.code;", + "if (mode === 'headers') {", + " pm.test('AS metadata is hidden in headers mode (404)', function () {", + " pm.expect(code, pm.response.text()).to.equal(404);", + " });", + " return;", + "}", + "pm.test('AS metadata is served (200)', function () { pm.expect(code).to.equal(200); });", + "var b = pm.response.json();", + "var issuer = pm.variables.get('mcp_issuer');", + "pm.test('issuer + endpoints are present', function () {", + " pm.expect(b.issuer).to.equal(issuer);", + " pm.expect(b.authorization_endpoint).to.equal(issuer + '/oauth2/authorize');", + " pm.expect(b.token_endpoint).to.equal(issuer + '/oauth2/token');", + " pm.expect(b.registration_endpoint).to.equal(issuer + '/oauth2/register');", + " pm.expect(b.jwks_uri).to.equal(issuer + '/.well-known/jwks.json');", + "});", + "pm.test('advertises code grant, PKCE S256, public clients', function () {", + " pm.expect(b.response_types_supported).to.include('code');", + " pm.expect(b.grant_types_supported).to.include('authorization_code');", + " pm.expect(b.grant_types_supported).to.include('refresh_token');", + " pm.expect(b.code_challenge_methods_supported).to.include('S256');", + " pm.expect(b.token_endpoint_auth_methods_supported).to.include('none');", + "});" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/.well-known/oauth-authorization-server", + "host": ["{{base_url}}"], + "path": [".well-known", "oauth-authorization-server"] + } + } + }, + { + "name": "JWKS", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "var code = pm.response.code;", + "if (mode === 'headers') {", + " pm.test('JWKS is hidden in headers mode (404)', function () {", + " pm.expect(code, pm.response.text()).to.equal(404);", + " });", + " return;", + "}", + "pm.test('JWKS is served (200)', function () { pm.expect(code).to.equal(200); });", + "var b = pm.response.json();", + "pm.test('exposes one RS256 signing key', function () {", + " pm.expect(b.keys).to.be.an('array').with.lengthOf(1);", + " var k = b.keys[0];", + " pm.expect(k.kty).to.equal('RSA');", + " pm.expect(k.alg).to.equal('RS256');", + " pm.expect(k.use).to.equal('sig');", + " pm.expect(k.kid).to.be.a('string').and.not.empty;", + " pm.expect(k.n).to.be.a('string').and.not.empty;", + " pm.expect(k.e).to.be.a('string').and.not.empty;", + "});" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/.well-known/jwks.json", + "host": ["{{base_url}}"], + "path": [".well-known", "jwks.json"] + } + } + } + ] + }, + { + "name": "MCP connect matrix", + "description": "POST a JSON-RPC initialize to /mcp with each credential. The auth gate runs on every /mcp request, so initialize alone proves accept/reject. Accepted is asserted as 'not 401'; rejected is asserted as exactly 401 (with WWW-Authenticate where discovery is enabled).", + "item": [ + { + "name": "Virtual key via x-bf-vk", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "var code = pm.response.code;", + "if (mode === 'oauth') {", + " pm.test('header VK is rejected in oauth-strict mode (401)', function () {", + " pm.expect(code, pm.response.text()).to.equal(401);", + " });", + " pm.test('401 advertises the protected-resource metadata', function () {", + " pm.expect(pm.response.headers.get('WWW-Authenticate')).to.be.a('string');", + " });", + "} else {", + " pm.test('header VK connects (not 401) — unchanged legacy behavior', function () {", + " pm.expect(code, pm.response.text()).to.not.equal(401);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { "key": "Content-Type", "value": "application/json" }, + { "key": "Accept", "value": "application/json, text/event-stream" }, + { "key": "x-bf-vk", "value": "{{vk_value}}" } + ], + "body": { + "mode": "raw", + "raw": "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"initialize\",\"params\":{\"protocolVersion\":\"2025-03-26\",\"capabilities\":{},\"clientInfo\":{\"name\":\"newman-mcp-auth\",\"version\":\"1.0.0\"}}}" + }, + "url": { "raw": "{{base_url}}/mcp", "host": ["{{base_url}}"], "path": ["mcp"] } + } + }, + { + "name": "Virtual key via Authorization Bearer", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "var code = pm.response.code;", + "if (mode === 'oauth') {", + " pm.test('Bearer VK is rejected in oauth-strict mode (401)', function () {", + " pm.expect(code, pm.response.text()).to.equal(401);", + " });", + "} else {", + " pm.test('Bearer VK connects (not 401) — unchanged legacy behavior', function () {", + " pm.expect(code, pm.response.text()).to.not.equal(401);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { "key": "Content-Type", "value": "application/json" }, + { "key": "Accept", "value": "application/json, text/event-stream" }, + { "key": "Authorization", "value": "Bearer {{vk_value}}" } + ], + "body": { + "mode": "raw", + "raw": "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"initialize\",\"params\":{\"protocolVersion\":\"2025-03-26\",\"capabilities\":{},\"clientInfo\":{\"name\":\"newman-mcp-auth\",\"version\":\"1.0.0\"}}}" + }, + "url": { "raw": "{{base_url}}/mcp", "host": ["{{base_url}}"], "path": ["mcp"] } + } + }, + { + "name": "Virtual key via x-api-key", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "var code = pm.response.code;", + "if (mode === 'oauth') {", + " pm.test('x-api-key VK is rejected in oauth-strict mode (401)', function () {", + " pm.expect(code, pm.response.text()).to.equal(401);", + " });", + "} else {", + " pm.test('x-api-key VK connects (not 401) — unchanged legacy behavior', function () {", + " pm.expect(code, pm.response.text()).to.not.equal(401);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { "key": "Content-Type", "value": "application/json" }, + { "key": "Accept", "value": "application/json, text/event-stream" }, + { "key": "x-api-key", "value": "{{vk_value}}" } + ], + "body": { + "mode": "raw", + "raw": "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"initialize\",\"params\":{\"protocolVersion\":\"2025-03-26\",\"capabilities\":{},\"clientInfo\":{\"name\":\"newman-mcp-auth\",\"version\":\"1.0.0\"}}}" + }, + "url": { "raw": "{{base_url}}/mcp", "host": ["{{base_url}}"], "path": ["mcp"] } + } + }, + { + "name": "No credentials (anonymous)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// enforce_auth_on_inference is off in the boot config, so anonymous /mcp", + "// access falls back to the global server in headers/both. oauth-strict", + "// still requires a JWT.", + "var mode = String(pm.variables.get('auth_mode') || '');", + "var code = pm.response.code;", + "if (mode === 'oauth') {", + " pm.test('anonymous is rejected in oauth-strict mode (401)', function () {", + " pm.expect(code, pm.response.text()).to.equal(401);", + " });", + " pm.test('401 advertises the protected-resource metadata', function () {", + " pm.expect(pm.response.headers.get('WWW-Authenticate')).to.be.a('string');", + " });", + "} else {", + " pm.test('anonymous connects when auth is not enforced (not 401)', function () {", + " pm.expect(code, pm.response.text()).to.not.equal(401);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { "key": "Content-Type", "value": "application/json" }, + { "key": "Accept", "value": "application/json, text/event-stream" } + ], + "body": { + "mode": "raw", + "raw": "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"initialize\",\"params\":{\"protocolVersion\":\"2025-03-26\",\"capabilities\":{},\"clientInfo\":{\"name\":\"newman-mcp-auth\",\"version\":\"1.0.0\"}}}" + }, + "url": { "raw": "{{base_url}}/mcp", "host": ["{{base_url}}"], "path": ["mcp"] } + } + }, + { + "name": "Inactive virtual key via x-bf-vk", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// A deactivated key must never connect, in any mode: headers/both reject", + "// it at the shared key-lookup chokepoint, oauth-strict rejects the header", + "// credential outright.", + "var code = pm.response.code;", + "pm.test('inactive virtual key is rejected (401)', function () {", + " pm.expect(code, pm.response.text()).to.equal(401);", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { "key": "Content-Type", "value": "application/json" }, + { "key": "Accept", "value": "application/json, text/event-stream" }, + { "key": "x-bf-vk", "value": "{{vk_inactive_value}}" } + ], + "body": { + "mode": "raw", + "raw": "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"initialize\",\"params\":{\"protocolVersion\":\"2025-03-26\",\"capabilities\":{},\"clientInfo\":{\"name\":\"newman-mcp-auth\",\"version\":\"1.0.0\"}}}" + }, + "url": { "raw": "{{base_url}}/mcp", "host": ["{{base_url}}"], "path": ["mcp"] } + } + }, + { + "name": "Opaque bearer token (not a Bifrost JWT)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// A JWT-shaped bearer that is not a real issued token. In headers mode the", + "// JWT path is off, so this is not treated as a credential and falls back to", + "// anonymous (auth not enforced). In both/oauth the token is verified and", + "// fails, yielding 401 with WWW-Authenticate.", + "var mode = String(pm.variables.get('auth_mode') || '');", + "var code = pm.response.code;", + "if (mode === 'headers') {", + " pm.test('JWT-shaped bearer is ignored in headers mode (not 401)', function () {", + " pm.expect(code, pm.response.text()).to.not.equal(401);", + " });", + "} else {", + " pm.test('invalid JWT is rejected (401)', function () {", + " pm.expect(code, pm.response.text()).to.equal(401);", + " });", + " pm.test('401 advertises the protected-resource metadata', function () {", + " pm.expect(pm.response.headers.get('WWW-Authenticate')).to.be.a('string');", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { "key": "Content-Type", "value": "application/json" }, + { "key": "Accept", "value": "application/json, text/event-stream" }, + { "key": "Authorization", "value": "Bearer eyJhbGciOiJSUzI1NiJ9.eyJzdWIiOiJmYWtlIn0.bm90LWEtcmVhbC1zaWduYXR1cmU" } + ], + "body": { + "mode": "raw", + "raw": "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"initialize\",\"params\":{\"protocolVersion\":\"2025-03-26\",\"capabilities\":{},\"clientInfo\":{\"name\":\"newman-mcp-auth\",\"version\":\"1.0.0\"}}}" + }, + "url": { "raw": "{{base_url}}/mcp", "host": ["{{base_url}}"], "path": ["mcp"] } + } + } + ] + }, + { + "name": "Config validation", + "description": "The /api/config endpoint guards the OAuth knobs. These PUTs are rejected (400) so they cause no state change, and the round-trip GET confirms the boot mode is persisted. Order-independent and safe to run before any runtime flip.", + "item": [ + { + "name": "Invalid mcp_server_auth_mode is rejected", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('unknown auth mode is rejected (400)', function () {", + " pm.expect(pm.response.code, pm.response.text()).to.equal(400);", + " pm.expect(pm.response.text()).to.include('mcp_server_auth_mode');", + "});" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [{ "key": "Content-Type", "value": "application/json" }], + "body": { + "mode": "raw", + "raw": "{\"client_config\":{\"log_retention_days\":30,\"mcp_server_auth_mode\":\"bogus\"}}" + }, + "url": { "raw": "{{base_url}}/api/config", "host": ["{{base_url}}"], "path": ["api", "config"] } + } + }, + { + "name": "oauth2_server_config rejected when mode is headers", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Explicit mode=headers in the payload makes the effective mode headers", + "// regardless of the boot mode, so the guard fires uniformly. The request", + "// fails validation, so it never changes the running mode.", + "pm.test('oauth2_server_config with headers mode is rejected (400)', function () {", + " pm.expect(pm.response.code, pm.response.text()).to.equal(400);", + " pm.expect(pm.response.text()).to.include('oauth2_server_config');", + "});" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [{ "key": "Content-Type", "value": "application/json" }], + "body": { + "mode": "raw", + "raw": "{\"client_config\":{\"log_retention_days\":30,\"mcp_server_auth_mode\":\"headers\",\"oauth2_server_config\":{\"issuer_url\":{\"value\":\"{{mcp_issuer}}\"},\"auth_code_ttl\":600,\"access_token_ttl\":600}}}" + }, + "url": { "raw": "{{base_url}}/api/config", "host": ["{{base_url}}"], "path": ["api", "config"] } + } + }, + { + "name": "Config round-trips the boot auth mode", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "pm.test('config readable (200)', function () { pm.expect(pm.response.code).to.equal(200); });", + "var cc = pm.response.json().client_config || {};", + "pm.test('mcp_server_auth_mode round-trips the boot mode', function () {", + " pm.expect(cc.mcp_server_auth_mode).to.equal(mode);", + "});", + "if (mode !== 'headers') {", + " pm.test('oauth2_server_config is persisted in both/oauth', function () {", + " pm.expect(cc.oauth2_server_config, JSON.stringify(cc.oauth2_server_config)).to.not.equal(undefined);", + " pm.expect(cc.oauth2_server_config).to.not.equal(null);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { "raw": "{{base_url}}/api/config", "host": ["{{base_url}}"], "path": ["api", "config"] } + } + } + ] + }, + { + "name": "Full OAuth flow (virtual-key identity)", + "description": "End-to-end issuance over HTTP: dynamic client registration, authorize (PKCE S256), consent bound to a virtual key, token exchange, connect to /mcp with the issued JWT, then refresh rotation and stolen-token family revocation. Runs only where issuance is enabled (both/oauth); in headers mode the steps are skipped. The consent API is reachable directly because admin auth is off in the boot config; the temp-token credential path is covered separately.", + "item": [ + { + "name": "Register client (DCR)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('issuance flow runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('client registered (201)', function () { pm.expect(pm.response.code).to.equal(201); });", + "var b = pm.response.json();", + "pm.collectionVariables.set('client_id', b.client_id);", + "pm.test('public client defaults', function () {", + " pm.expect(b.client_id).to.be.a('string').and.not.empty;", + " pm.expect(b.token_endpoint_auth_method).to.equal('none');", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [{ "key": "Content-Type", "value": "application/json" }], + "body": { + "mode": "raw", + "raw": "{\"client_name\":\"newman-flow-client\",\"redirect_uris\":[\"http://127.0.0.1:9999/cb\"],\"grant_types\":[\"authorization_code\",\"refresh_token\"]}" + }, + "url": { "raw": "{{base_url}}/oauth2/register", "host": ["{{base_url}}"], "path": ["oauth2", "register"] } + } + }, + { + "name": "Authorize (PKCE S256)", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "// Fresh PKCE pair per flow. challenge = base64url(sha256(verifier)).", + "var verifier = CryptoJS.lib.WordArray.random(32).toString(CryptoJS.enc.Hex);", + "var challenge = CryptoJS.enc.Base64.stringify(CryptoJS.SHA256(verifier))", + " .replace(/\\+/g, '-').replace(/\\//g, '_').replace(/=+$/, '');", + "pm.collectionVariables.set('code_verifier', verifier);", + "pm.collectionVariables.set('code_challenge', challenge);" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('issuance flow runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('authorize redirects to consent (302)', function () { pm.expect(pm.response.code).to.equal(302); });", + "var loc = pm.response.headers.get('Location') || '';", + "pm.test('redirect targets the consent page', function () { pm.expect(loc).to.include('/oauth/consent'); });", + "var flow = loc.match(/[?&]flow=([^&#]+)/);", + "pm.expect(flow, 'flow id in redirect').to.not.equal(null);", + "pm.collectionVariables.set('flow_id', decodeURIComponent(flow[1]));" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/oauth2/authorize?response_type=code&client_id={{client_id}}&redirect_uri=http://127.0.0.1:9999/cb&code_challenge={{code_challenge}}&code_challenge_method=S256&resource={{mcp_issuer}}/mcp&state=xyz&scope=mcp", + "host": ["{{base_url}}"], + "path": ["oauth2", "authorize"], + "query": [ + { "key": "response_type", "value": "code" }, + { "key": "client_id", "value": "{{client_id}}" }, + { "key": "redirect_uri", "value": "http://127.0.0.1:9999/cb" }, + { "key": "code_challenge", "value": "{{code_challenge}}" }, + { "key": "code_challenge_method", "value": "S256" }, + { "key": "resource", "value": "{{mcp_issuer}}/mcp" }, + { "key": "state", "value": "xyz" }, + { "key": "scope", "value": "mcp" } + ] + } + } + }, + { + "name": "Consent as virtual key", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('issuance flow runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('consent succeeds (200)', function () { pm.expect(pm.response.code, pm.response.text()).to.equal(200); });", + "var url = pm.response.json().redirect_url || '';", + "var m = url.match(/[?&]code=([^&]+)/);", + "pm.expect(m, 'authorization code in redirect_url').to.not.equal(null);", + "pm.collectionVariables.set('auth_code', decodeURIComponent(m[1]));" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [{ "key": "Content-Type", "value": "application/json" }], + "body": { + "mode": "raw", + "raw": "{\"mode\":\"vk\",\"value\":\"{{vk_value}}\"}" + }, + "url": { + "raw": "{{base_url}}/api/oauth2/consent/flows/{{flow_id}}", + "host": ["{{base_url}}"], + "path": ["api", "oauth2", "consent", "flows", "{{flow_id}}"] + } + } + }, + { + "name": "Token exchange (authorization_code)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('issuance flow runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('token issued (200)', function () { pm.expect(pm.response.code, pm.response.text()).to.equal(200); });", + "var b = pm.response.json();", + "pm.test('bearer access + refresh returned', function () {", + " pm.expect(b.token_type).to.equal('Bearer');", + " pm.expect(b.access_token).to.be.a('string').and.not.empty;", + " pm.expect(b.refresh_token).to.be.a('string').and.not.empty;", + "});", + "pm.test('expires_in reflects the configured access_token_ttl', function () {", + " pm.expect(b.expires_in).to.equal(600);", + "});", + "pm.collectionVariables.set('access_token', b.access_token);", + "pm.collectionVariables.set('refresh_token', b.refresh_token);" + ] + } + } + ], + "request": { + "method": "POST", + "header": [{ "key": "Content-Type", "value": "application/x-www-form-urlencoded" }], + "body": { + "mode": "urlencoded", + "urlencoded": [ + { "key": "grant_type", "value": "authorization_code" }, + { "key": "code", "value": "{{auth_code}}" }, + { "key": "code_verifier", "value": "{{code_verifier}}" }, + { "key": "client_id", "value": "{{client_id}}" }, + { "key": "redirect_uri", "value": "http://127.0.0.1:9999/cb" }, + { "key": "resource", "value": "{{mcp_issuer}}/mcp" } + ] + }, + "url": { "raw": "{{base_url}}/oauth2/token", "host": ["{{base_url}}"], "path": ["oauth2", "token"] } + } + }, + { + "name": "Connect to /mcp with issued JWT", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('issuance flow runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('issued JWT connects to /mcp (not 401)', function () {", + " pm.expect(pm.response.code, pm.response.text()).to.not.equal(401);", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { "key": "Content-Type", "value": "application/json" }, + { "key": "Accept", "value": "application/json, text/event-stream" }, + { "key": "Authorization", "value": "Bearer {{access_token}}" } + ], + "body": { + "mode": "raw", + "raw": "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"initialize\",\"params\":{\"protocolVersion\":\"2025-03-26\",\"capabilities\":{},\"clientInfo\":{\"name\":\"newman-mcp-auth\",\"version\":\"1.0.0\"}}}" + }, + "url": { "raw": "{{base_url}}/mcp", "host": ["{{base_url}}"], "path": ["mcp"] } + } + }, + { + "name": "Refresh rotation", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('issuance flow runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('refresh succeeds (200)', function () { pm.expect(pm.response.code, pm.response.text()).to.equal(200); });", + "var b = pm.response.json();", + "pm.test('rotation returns a new refresh token', function () {", + " pm.expect(b.refresh_token).to.be.a('string').and.not.empty;", + " pm.expect(b.refresh_token).to.not.equal(pm.collectionVariables.get('refresh_token'));", + "});", + "// Keep the prior (now-rotated) refresh token to prove replay revokes the family.", + "pm.collectionVariables.set('old_refresh_token', pm.collectionVariables.get('refresh_token'));", + "pm.collectionVariables.set('refresh_token', b.refresh_token);" + ] + } + } + ], + "request": { + "method": "POST", + "header": [{ "key": "Content-Type", "value": "application/x-www-form-urlencoded" }], + "body": { + "mode": "urlencoded", + "urlencoded": [ + { "key": "grant_type", "value": "refresh_token" }, + { "key": "refresh_token", "value": "{{refresh_token}}" }, + { "key": "client_id", "value": "{{client_id}}" } + ] + }, + "url": { "raw": "{{base_url}}/oauth2/token", "host": ["{{base_url}}"], "path": ["oauth2", "token"] } + } + }, + { + "name": "Replay rotated refresh token", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('issuance flow runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('replaying a rotated refresh token is rejected (400)', function () {", + " pm.expect(pm.response.code, pm.response.text()).to.equal(400);", + " pm.expect(pm.response.text()).to.include('invalid_grant');", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [{ "key": "Content-Type", "value": "application/x-www-form-urlencoded" }], + "body": { + "mode": "urlencoded", + "urlencoded": [ + { "key": "grant_type", "value": "refresh_token" }, + { "key": "refresh_token", "value": "{{old_refresh_token}}" }, + { "key": "client_id", "value": "{{client_id}}" } + ] + }, + "url": { "raw": "{{base_url}}/oauth2/token", "host": ["{{base_url}}"], "path": ["oauth2", "token"] } + } + }, + { + "name": "Family revoked after replay", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('issuance flow runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "// The replay tripped stolen-token detection, so the latest (legitimate)", + "// refresh token in the same family is now revoked too.", + "pm.test('post-replay the live refresh token is also revoked (400)', function () {", + " pm.expect(pm.response.code, pm.response.text()).to.equal(400);", + " pm.expect(pm.response.text()).to.include('invalid_grant');", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [{ "key": "Content-Type", "value": "application/x-www-form-urlencoded" }], + "body": { + "mode": "urlencoded", + "urlencoded": [ + { "key": "grant_type", "value": "refresh_token" }, + { "key": "refresh_token", "value": "{{refresh_token}}" }, + { "key": "client_id", "value": "{{client_id}}" } + ] + }, + "url": { "raw": "{{base_url}}/oauth2/token", "host": ["{{base_url}}"], "path": ["oauth2", "token"] } + } + } + ] + }, + { + "name": "Revocation window", + "description": "Revoking a grant stops the refresh token immediately, but the already-issued short-lived access token keeps working on /mcp until it expires. Runs a fresh grant (both/oauth only), revokes it via the management API, then asserts both halves of the documented window.", + "item": [ + { + "name": "Fresh authorize for revocation test", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "var verifier = CryptoJS.lib.WordArray.random(32).toString(CryptoJS.enc.Hex);", + "var challenge = CryptoJS.enc.Base64.stringify(CryptoJS.SHA256(verifier))", + " .replace(/\\+/g, '-').replace(/\\//g, '_').replace(/=+$/, '');", + "pm.collectionVariables.set('rev_verifier', verifier);", + "pm.collectionVariables.set('rev_challenge', challenge);", + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { return; }", + "// Snapshot the active grant ids BEFORE this flow creates a new one, so the", + "// new grant is isolated later by set-difference rather than by list order.", + "pm.sendRequest({ url: pm.variables.get('base_url') + '/api/oauth2/sessions', method: 'GET' }, function (err, res) {", + " var ids = [];", + " if (!err && res && res.code === 200) { ids = (res.json().sessions || []).map(function (s) { return s.id; }); }", + " pm.collectionVariables.set('rev_pre_ids', JSON.stringify(ids));", + "});" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('revocation window runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('authorize redirects (302)', function () { pm.expect(pm.response.code).to.equal(302); });", + "var loc = pm.response.headers.get('Location') || '';", + "var flow = loc.match(/[?&]flow=([^&#]+)/);", + "pm.expect(flow, 'flow id').to.not.equal(null);", + "pm.collectionVariables.set('rev_flow_id', decodeURIComponent(flow[1]));" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/oauth2/authorize?response_type=code&client_id={{client_id}}&redirect_uri=http://127.0.0.1:9999/cb&code_challenge={{rev_challenge}}&code_challenge_method=S256&resource={{mcp_issuer}}/mcp&state=rev&scope=mcp", + "host": ["{{base_url}}"], + "path": ["oauth2", "authorize"], + "query": [ + { "key": "response_type", "value": "code" }, + { "key": "client_id", "value": "{{client_id}}" }, + { "key": "redirect_uri", "value": "http://127.0.0.1:9999/cb" }, + { "key": "code_challenge", "value": "{{rev_challenge}}" }, + { "key": "code_challenge_method", "value": "S256" }, + { "key": "resource", "value": "{{mcp_issuer}}/mcp" }, + { "key": "state", "value": "rev" }, + { "key": "scope", "value": "mcp" } + ] + } + } + }, + { + "name": "Consent + token for revocation test", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('revocation window runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('consent succeeds (200)', function () { pm.expect(pm.response.code, pm.response.text()).to.equal(200); });", + "var url = pm.response.json().redirect_url || '';", + "var m = url.match(/[?&]code=([^&]+)/);", + "pm.expect(m, 'code').to.not.equal(null);", + "pm.collectionVariables.set('rev_code', decodeURIComponent(m[1]));" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [{ "key": "Content-Type", "value": "application/json" }], + "body": { "mode": "raw", "raw": "{\"mode\":\"vk\",\"value\":\"{{vk_value}}\"}" }, + "url": { + "raw": "{{base_url}}/api/oauth2/consent/flows/{{rev_flow_id}}", + "host": ["{{base_url}}"], + "path": ["api", "oauth2", "consent", "flows", "{{rev_flow_id}}"] + } + } + }, + { + "name": "Exchange code for revocation test", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('revocation window runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('token issued (200)', function () { pm.expect(pm.response.code, pm.response.text()).to.equal(200); });", + "var b = pm.response.json();", + "pm.collectionVariables.set('rev_access_token', b.access_token);", + "pm.collectionVariables.set('rev_refresh_token', b.refresh_token);" + ] + } + } + ], + "request": { + "method": "POST", + "header": [{ "key": "Content-Type", "value": "application/x-www-form-urlencoded" }], + "body": { + "mode": "urlencoded", + "urlencoded": [ + { "key": "grant_type", "value": "authorization_code" }, + { "key": "code", "value": "{{rev_code}}" }, + { "key": "code_verifier", "value": "{{rev_verifier}}" }, + { "key": "client_id", "value": "{{client_id}}" }, + { "key": "redirect_uri", "value": "http://127.0.0.1:9999/cb" }, + { "key": "resource", "value": "{{mcp_issuer}}/mcp" } + ] + }, + "url": { "raw": "{{base_url}}/oauth2/token", "host": ["{{base_url}}"], "path": ["oauth2", "token"] } + } + }, + { + "name": "List grants and pick the new one", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('revocation window runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('grants list returns (200)', function () { pm.expect(pm.response.code, pm.response.text()).to.equal(200); });", + "var sessions = (pm.response.json().sessions) || [];", + "var pre = JSON.parse(pm.collectionVariables.get('rev_pre_ids') || '[]');", + "// Isolate the grant created by this flow as the one absent from the", + "// pre-flow snapshot — independent of list ordering or other grants.", + "var fresh = sessions.filter(function (s) { return pre.indexOf(s.id) === -1; });", + "pm.test('exactly one new grant was created by this flow', function () {", + " pm.expect(fresh.length, JSON.stringify(fresh)).to.equal(1);", + "});", + "if (fresh.length === 1) { pm.collectionVariables.set('rev_grant_id', fresh[0].id); }" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { "raw": "{{base_url}}/api/oauth2/sessions", "host": ["{{base_url}}"], "path": ["api", "oauth2", "sessions"] } + } + }, + { + "name": "Revoke the grant", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('revocation window runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('grant revoked (204)', function () { pm.expect(pm.response.code, pm.response.text()).to.equal(204); });" + ] + } + } + ], + "request": { + "method": "DELETE", + "header": [], + "url": { + "raw": "{{base_url}}/api/oauth2/sessions/{{rev_grant_id}}", + "host": ["{{base_url}}"], + "path": ["api", "oauth2", "sessions", "{{rev_grant_id}}"] + } + } + }, + { + "name": "Refresh after revoke is rejected", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('revocation window runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('refresh after revoke is rejected (400)', function () {", + " pm.expect(pm.response.code, pm.response.text()).to.equal(400);", + " pm.expect(pm.response.text()).to.include('invalid_grant');", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [{ "key": "Content-Type", "value": "application/x-www-form-urlencoded" }], + "body": { + "mode": "urlencoded", + "urlencoded": [ + { "key": "grant_type", "value": "refresh_token" }, + { "key": "refresh_token", "value": "{{rev_refresh_token}}" }, + { "key": "client_id", "value": "{{client_id}}" } + ] + }, + "url": { "raw": "{{base_url}}/oauth2/token", "host": ["{{base_url}}"], "path": ["oauth2", "token"] } + } + }, + { + "name": "Issued access token still connects within the window", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('revocation window runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "// The short-lived access token is a stateless JWT, so it keeps working", + "// until exp even though the refresh token is revoked — the documented window.", + "pm.test('already-issued access token still connects (not 401)', function () {", + " pm.expect(pm.response.code, pm.response.text()).to.not.equal(401);", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { "key": "Content-Type", "value": "application/json" }, + { "key": "Accept", "value": "application/json, text/event-stream" }, + { "key": "Authorization", "value": "Bearer {{rev_access_token}}" } + ], + "body": { + "mode": "raw", + "raw": "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"initialize\",\"params\":{\"protocolVersion\":\"2025-03-26\",\"capabilities\":{},\"clientInfo\":{\"name\":\"newman-mcp-auth\",\"version\":\"1.0.0\"}}}" + }, + "url": { "raw": "{{base_url}}/mcp", "host": ["{{base_url}}"], "path": ["mcp"] } + } + } + ] + }, + { + "name": "Full OAuth flow (session identity)", + "description": "Session-mode issuance: the consent server mints an opaque session identity (never client-asserted), the token connects to /mcp, and then enabling enforce_auth_on_inference at runtime makes that same session token unacceptable. Runs only where issuance is enabled (both/oauth) and reuses the client registered by the virtual-key flow. The enforce flip is the last meaningful step in the boot, so it does not affect earlier folders.", + "item": [ + { + "name": "Authorize for session flow", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "var verifier = CryptoJS.lib.WordArray.random(32).toString(CryptoJS.enc.Hex);", + "var challenge = CryptoJS.enc.Base64.stringify(CryptoJS.SHA256(verifier))", + " .replace(/\\+/g, '-').replace(/\\//g, '_').replace(/=+$/, '');", + "pm.collectionVariables.set('sess_verifier', verifier);", + "pm.collectionVariables.set('sess_challenge', challenge);" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('session flow runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('authorize redirects (302)', function () { pm.expect(pm.response.code).to.equal(302); });", + "var loc = pm.response.headers.get('Location') || '';", + "var flow = loc.match(/[?&]flow=([^&#]+)/);", + "pm.expect(flow, 'flow id').to.not.equal(null);", + "pm.collectionVariables.set('sess_flow_id', decodeURIComponent(flow[1]));" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/oauth2/authorize?response_type=code&client_id={{client_id}}&redirect_uri=http://127.0.0.1:9999/cb&code_challenge={{sess_challenge}}&code_challenge_method=S256&resource={{mcp_issuer}}/mcp&state=sess&scope=mcp", + "host": ["{{base_url}}"], + "path": ["oauth2", "authorize"], + "query": [ + { "key": "response_type", "value": "code" }, + { "key": "client_id", "value": "{{client_id}}" }, + { "key": "redirect_uri", "value": "http://127.0.0.1:9999/cb" }, + { "key": "code_challenge", "value": "{{sess_challenge}}" }, + { "key": "code_challenge_method", "value": "S256" }, + { "key": "resource", "value": "{{mcp_issuer}}/mcp" }, + { "key": "state", "value": "sess" }, + { "key": "scope", "value": "mcp" } + ] + } + } + }, + { + "name": "Consent as session", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('session flow runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('consent succeeds (200)', function () { pm.expect(pm.response.code, pm.response.text()).to.equal(200); });", + "var url = pm.response.json().redirect_url || '';", + "var m = url.match(/[?&]code=([^&]+)/);", + "pm.expect(m, 'code').to.not.equal(null);", + "pm.collectionVariables.set('sess_code', decodeURIComponent(m[1]));" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [{ "key": "Content-Type", "value": "application/json" }], + "body": { "mode": "raw", "raw": "{\"mode\":\"session\"}" }, + "url": { + "raw": "{{base_url}}/api/oauth2/consent/flows/{{sess_flow_id}}", + "host": ["{{base_url}}"], + "path": ["api", "oauth2", "consent", "flows", "{{sess_flow_id}}"] + } + } + }, + { + "name": "Token exchange for session flow", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('session flow runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('token issued (200)', function () { pm.expect(pm.response.code, pm.response.text()).to.equal(200); });", + "pm.collectionVariables.set('sess_access_token', pm.response.json().access_token);" + ] + } + } + ], + "request": { + "method": "POST", + "header": [{ "key": "Content-Type", "value": "application/x-www-form-urlencoded" }], + "body": { + "mode": "urlencoded", + "urlencoded": [ + { "key": "grant_type", "value": "authorization_code" }, + { "key": "code", "value": "{{sess_code}}" }, + { "key": "code_verifier", "value": "{{sess_verifier}}" }, + { "key": "client_id", "value": "{{client_id}}" }, + { "key": "redirect_uri", "value": "http://127.0.0.1:9999/cb" }, + { "key": "resource", "value": "{{mcp_issuer}}/mcp" } + ] + }, + "url": { "raw": "{{base_url}}/oauth2/token", "host": ["{{base_url}}"], "path": ["oauth2", "token"] } + } + }, + { + "name": "Session JWT connects while auth is not enforced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('session flow runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('session-mode JWT connects (not 401)', function () {", + " pm.expect(pm.response.code, pm.response.text()).to.not.equal(401);", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { "key": "Content-Type", "value": "application/json" }, + { "key": "Accept", "value": "application/json, text/event-stream" }, + { "key": "Authorization", "value": "Bearer {{sess_access_token}}" } + ], + "body": { + "mode": "raw", + "raw": "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"initialize\",\"params\":{\"protocolVersion\":\"2025-03-26\",\"capabilities\":{},\"clientInfo\":{\"name\":\"newman-mcp-auth\",\"version\":\"1.0.0\"}}}" + }, + "url": { "raw": "{{base_url}}/mcp", "host": ["{{base_url}}"], "path": ["mcp"] } + } + }, + { + "name": "Enable enforce_auth_on_inference", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('session flow runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('enforce_auth enabled (2xx)', function () { pm.expect(pm.response.code, pm.response.text()).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [{ "key": "Content-Type", "value": "application/json" }], + "body": { + "mode": "raw", + "raw": "{\"client_config\":{\"log_retention_days\":30,\"enforce_auth_on_inference\":true}}" + }, + "url": { "raw": "{{base_url}}/api/config", "host": ["{{base_url}}"], "path": ["api", "config"] } + } + }, + { + "name": "Session JWT rejected once auth is enforced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode === 'headers') { pm.test('session flow runs in both/oauth modes', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('session-mode JWT is rejected when auth is enforced (401)', function () {", + " pm.expect(pm.response.code, pm.response.text()).to.equal(401);", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { "key": "Content-Type", "value": "application/json" }, + { "key": "Accept", "value": "application/json, text/event-stream" }, + { "key": "Authorization", "value": "Bearer {{sess_access_token}}" } + ], + "body": { + "mode": "raw", + "raw": "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"initialize\",\"params\":{\"protocolVersion\":\"2025-03-26\",\"capabilities\":{},\"clientInfo\":{\"name\":\"newman-mcp-auth\",\"version\":\"1.0.0\"}}}" + }, + "url": { "raw": "{{base_url}}/mcp", "host": ["{{base_url}}"], "path": ["mcp"] } + } + } + ] + }, + { + "name": "Runtime config flip (headers to both)", + "description": "Proves the legacy-to-mixed upgrade at runtime: starting from headers mode, PUT /api/config to switch to both, then confirm discovery comes alive AND a header-VK /mcp connect still works (no regression). Runs only in the headers boot; in both/oauth boots it is a no-op. The PUT is a minimal partial — only the fields under test plus log_retention_days, which the endpoint validates unconditionally.", + "item": [ + { + "name": "Flip mcp_server_auth_mode to both", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode !== 'headers') { pm.test('runtime flip is exercised from the headers boot', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('config update accepted (2xx)', function () { pm.expect(pm.response.code, pm.response.text()).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [{ "key": "Content-Type", "value": "application/json" }], + "body": { + "mode": "raw", + "raw": "{\"client_config\":{\"log_retention_days\":30,\"mcp_server_auth_mode\":\"both\",\"oauth2_server_config\":{\"issuer_url\":{\"value\":\"{{mcp_issuer}}\"},\"auth_code_ttl\":600,\"access_token_ttl\":600}}}" + }, + "url": { "raw": "{{base_url}}/api/config", "host": ["{{base_url}}"], "path": ["api", "config"] } + } + }, + { + "name": "Discovery is now live after the flip", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode !== 'headers') { pm.test('runtime flip is exercised from the headers boot', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('protected-resource metadata is now served (200)', function () {", + " pm.expect(pm.response.code, pm.response.text()).to.equal(200);", + "});" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/.well-known/oauth-protected-resource", + "host": ["{{base_url}}"], + "path": [".well-known", "oauth-protected-resource"] + } + } + }, + { + "name": "Header VK still connects after the flip", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var mode = String(pm.variables.get('auth_mode') || '');", + "if (mode !== 'headers') { pm.test('runtime flip is exercised from the headers boot', function () { pm.expect(true).to.be.true; }); return; }", + "pm.test('header VK connect is unaffected by the upgrade (not 401)', function () {", + " pm.expect(pm.response.code, pm.response.text()).to.not.equal(401);", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { "key": "Content-Type", "value": "application/json" }, + { "key": "Accept", "value": "application/json, text/event-stream" }, + { "key": "x-bf-vk", "value": "{{vk_value}}" } + ], + "body": { + "mode": "raw", + "raw": "{\"jsonrpc\":\"2.0\",\"id\":1,\"method\":\"initialize\",\"params\":{\"protocolVersion\":\"2025-03-26\",\"capabilities\":{},\"clientInfo\":{\"name\":\"newman-mcp-auth\",\"version\":\"1.0.0\"}}}" + }, + "url": { "raw": "{{base_url}}/mcp", "host": ["{{base_url}}"], "path": ["mcp"] } + } + } + ] + } + ] +} diff --git a/tests/e2e/api/collections/bifrost-v1-vk-expiry.postman_collection.json b/tests/e2e/api/collections/bifrost-v1-vk-expiry.postman_collection.json new file mode 100644 index 00000000000..2c3e12ec532 --- /dev/null +++ b/tests/e2e/api/collections/bifrost-v1-vk-expiry.postman_collection.json @@ -0,0 +1,1899 @@ +{ + "info": { + "name": "Bifrost V1 - Virtual Key Expiry", + "description": "Virtual key expiry tests. Covers create/update validation (future-only timestamps, RFC3339 parsing, empty-string clear, omit-leaves-unchanged) and runtime enforcement (unexpired passes, expired blocks with 403 virtual_key_blocked, inactive wins before expired, clearing expiry restores access). Self-provisions VKs and cleans up.", + "schema": "https://schema.getpostman.com/json/collection/v2.1.0/collection.json" + }, + "variable": [ + { + "key": "base_url", + "value": "http://localhost:8080", + "type": "string" + }, + { + "key": "provider", + "value": "openai", + "type": "string" + }, + { + "key": "chat_model", + "value": "gpt-4o", + "type": "string" + }, + { + "key": "vk_id", + "value": "", + "type": "string" + }, + { + "key": "vk_value", + "value": "", + "type": "string" + }, + { + "key": "vk2_id", + "value": "", + "type": "string" + }, + { + "key": "mcp_client_name", + "value": "", + "type": "string" + }, + { + "key": "mcp_tool_name", + "value": "", + "type": "string" + }, + { + "key": "mcp_candidates", + "value": "", + "type": "string" + }, + { + "key": "short_expiry_epoch", + "value": "", + "type": "string" + } + ], + "item": [ + { + "name": "Setup", + "item": [ + { + "name": "Create VK without expiry", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "var timestamp = Date.now();", + "var uniqueName = 'VK Expiry Test ' + timestamp;", + "var provider = pm.variables.get('provider') || 'openai';", + "pm.request.body.raw = JSON.stringify({name: uniqueName, provider_configs: [{provider: provider, weight: 1.0, allowed_models: ['*'], key_ids: ['*']}]});" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var code = pm.response.code;", + "pm.test('Create VK returns 200 or 201', function() { pm.expect(code).to.be.oneOf([200, 201]); });", + "if (code === 200 || code === 201) {", + " var jsonData = pm.response.json();", + " var vk = (jsonData && (jsonData.virtual_key || jsonData)) || null;", + " pm.test('VK has id and value', function() {", + " pm.expect(vk).to.be.an('object');", + " pm.expect(vk.id).to.be.a('string').and.not.be.empty;", + " pm.expect(vk.value).to.be.a('string').and.not.be.empty;", + " });", + " pm.test('VK without expiry has no expires_at', function() {", + " pm.expect(vk.expires_at == null).to.be.true;", + " });", + " pm.collectionVariables.set('vk_id', vk.id);", + " pm.collectionVariables.set('vk_value', vk.value);", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\"name\": \"VK Expiry Test\"}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys" + ] + } + } + }, + { + "name": "Discover MCP tool candidates for enforcement tests", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Executable tool names are registered as '-'. Collect", + "// candidates from connected clients, restricted to read-only-looking,", + "// argument-free-callable tools so probing them is side-effect free.", + "// Per-user-auth clients fail before the governance gate, so candidates", + "// are verified with a live probe in the next request.", + "pm.test('List MCP clients returns 200', function() { pm.expect(pm.response.code).to.equal(200); });", + "var jsonData = pm.response.json();", + "var clients = (jsonData && jsonData.clients) || [];", + "var SAFE_TOOL = /(echo|greet|public|info|ping|health|status)/i;", + "var candidates = [];", + "clients.forEach(function(cl) {", + " if (cl.state !== 'connected' || !cl.config || !cl.config.name) return;", + " (cl.tools || []).forEach(function(t) {", + " if (t.name && SAFE_TOOL.test(t.name) && candidates.length < 5) {", + " candidates.push({client: cl.config.name, tool: cl.config.name + '-' + t.name});", + " }", + " });", + "});", + "pm.collectionVariables.set('mcp_candidates', JSON.stringify(candidates));", + "if (!candidates.length) {", + " pm.test('Skipped: no connected MCP client with a safe probe tool', function() { pm.expect(true).to.be.true; });", + "} else {", + " console.log('MCP candidates: ' + candidates.map(function(c) { return c.tool; }).join(', '));", + "}" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/api/mcp/clients", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "mcp", + "clients" + ] + } + } + }, + { + "name": "Grant VK access to MCP candidate clients", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "var candidates = JSON.parse(pm.collectionVariables.get('mcp_candidates') || '[]');", + "var clientNames = candidates.map(function(c) { return c.client; }).filter(function(v, i, a) { return a.indexOf(v) === i; });", + "// No usable MCP client: turn this into a harmless no-op update.", + "if (!clientNames.length) {", + " pm.request.body.raw = JSON.stringify({description: 'no MCP client available'});", + "} else {", + " pm.request.body.raw = JSON.stringify({mcp_configs: clientNames.map(function(n) { return {mcp_client_name: n, tools_to_execute: ['*']}; })});", + "}" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Grant MCP access returns 2xx', function() { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "Probe MCP candidates and pick a working tool", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// A candidate only counts if a real execution with the fresh VK returns", + "// 200 \u2014 this filters out per-user-auth clients, whose connection fails", + "// before the governance gate the expiry tests need to reach.", + "var candidates = JSON.parse(pm.collectionVariables.get('mcp_candidates') || '[]');", + "pm.collectionVariables.set('mcp_tool_name', '');", + "if (!candidates.length) {", + " pm.test('Skipped: no MCP candidates to probe', function() { pm.expect(true).to.be.true; });", + "} else {", + " var baseUrl = pm.variables.get('base_url');", + " var vkValue = pm.collectionVariables.get('vk_value');", + " function probe(i) {", + " if (i >= candidates.length) {", + " pm.test('Skipped: no MCP candidate executed successfully', function() { pm.expect(true).to.be.true; });", + " return;", + " }", + " pm.sendRequest({", + " url: baseUrl + '/v1/mcp/tool/execute',", + " method: 'POST',", + " header: {'Content-Type': 'application/json', 'x-bf-vk': vkValue},", + " body: {mode: 'raw', raw: JSON.stringify({id: 'probe_' + i, type: 'function', function: {name: candidates[i].tool, arguments: '{}'}})}", + " }, function(err, res) {", + " if (!err && res.code === 200) {", + " pm.collectionVariables.set('mcp_tool_name', candidates[i].tool);", + " pm.test('Selected MCP tool for enforcement tests: ' + candidates[i].tool, function() { pm.expect(true).to.be.true; });", + " } else {", + " probe(i + 1);", + " }", + " });", + " }", + " probe(0);", + "}" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/api/mcp/clients", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "mcp", + "clients" + ] + } + } + } + ] + }, + { + "name": "Create Validation", + "item": [ + { + "name": "Create VK with past expiry is rejected", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "var uniqueName = 'VK Expiry Past ' + Date.now();", + "var pastExpiry = new Date(Date.now() - 3600000).toISOString();", + "pm.request.body.raw = JSON.stringify({name: uniqueName, expires_at: pastExpiry});" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Create with past expiry returns 400', function() { pm.expect(pm.response.code).to.equal(400); });", + "pm.test('Error mentions future timestamp', function() { pm.expect(pm.response.text()).to.include('future'); });" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\"name\": \"VK Expiry Past\"}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys" + ] + } + } + }, + { + "name": "Create VK with future expiry persists it", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "var uniqueName = 'VK Expiry Future ' + Date.now();", + "var futureExpiry = new Date(Date.now() + 3600000).toISOString();", + "pm.request.body.raw = JSON.stringify({name: uniqueName, expires_at: futureExpiry});" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var code = pm.response.code;", + "pm.test('Create with future expiry returns 200 or 201', function() { pm.expect(code).to.be.oneOf([200, 201]); });", + "if (code === 200 || code === 201) {", + " var jsonData = pm.response.json();", + " var vk = (jsonData && (jsonData.virtual_key || jsonData)) || null;", + " pm.test('Response echoes expires_at about 1h from now', function() {", + " pm.expect(vk).to.be.an('object');", + " pm.expect(vk.expires_at).to.be.a('string').and.not.be.empty;", + " var exp = new Date(vk.expires_at).getTime();", + " pm.expect(exp).to.be.greaterThan(Date.now() + 50 * 60000);", + " pm.expect(exp).to.be.lessThan(Date.now() + 70 * 60000);", + " });", + " if (vk && vk.id) { pm.collectionVariables.set('vk2_id', vk.id); }", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\"name\": \"VK Expiry Future\"}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys" + ] + } + } + } + ] + }, + { + "name": "Update Semantics", + "item": [ + { + "name": "Set future expiry via update", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "var futureExpiry = new Date(Date.now() + 3600000).toISOString();", + "pm.request.body.raw = JSON.stringify({expires_at: futureExpiry});" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Set expiry returns 2xx', function() { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "GET reflects the new expiry", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('GET VK returns 200', function() { pm.expect(pm.response.code).to.equal(200); });", + "var jsonData = pm.response.json();", + "var vk = (jsonData && (jsonData.virtual_key || jsonData)) || null;", + "pm.test('expires_at is about 1h from now', function() {", + " pm.expect(vk).to.be.an('object');", + " pm.expect(vk.expires_at).to.be.a('string').and.not.be.empty;", + " var exp = new Date(vk.expires_at).getTime();", + " pm.expect(exp).to.be.greaterThan(Date.now() + 50 * 60000);", + " pm.expect(exp).to.be.lessThan(Date.now() + 70 * 60000);", + "});" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "List endpoint includes expires_at", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('List VKs returns 200', function() { pm.expect(pm.response.code).to.equal(200); });", + "var jsonData = pm.response.json();", + "var list = (jsonData && jsonData.virtual_keys) || [];", + "var mine = list.filter(function(v) { return v.id === pm.collectionVariables.get('vk_id'); })[0];", + "if (mine) {", + " pm.test('Listed VK carries its expires_at', function() {", + " pm.expect(mine.expires_at).to.be.a('string').and.not.be.empty;", + " pm.expect(new Date(mine.expires_at).getTime()).to.be.greaterThan(Date.now());", + " });", + "} else {", + " pm.test('Skipped: VK not on this list page', function() { pm.expect(true).to.be.true; });", + "}" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys?limit=500&offset=0", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys" + ], + "query": [ + { + "key": "limit", + "value": "500" + }, + { + "key": "offset", + "value": "0" + } + ] + } + } + }, + { + "name": "Update without expires_at leaves expiry unchanged", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Update without expires_at returns 2xx', function() { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\"description\": \"expiry untouched\"}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "GET still has the expiry", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('GET VK returns 200', function() { pm.expect(pm.response.code).to.equal(200); });", + "var jsonData = pm.response.json();", + "var vk = (jsonData && (jsonData.virtual_key || jsonData)) || null;", + "pm.test('expires_at survives an unrelated update', function() {", + " pm.expect(vk).to.be.an('object');", + " pm.expect(vk.expires_at).to.be.a('string').and.not.be.empty;", + " pm.expect(new Date(vk.expires_at).getTime()).to.be.greaterThan(Date.now());", + "});" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "Extend expiry to 2h", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "var extended = new Date(Date.now() + 7200000).toISOString();", + "pm.request.body.raw = JSON.stringify({expires_at: extended});" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Extend expiry returns 2xx', function() { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "GET reflects the extended expiry", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('GET VK returns 200', function() { pm.expect(pm.response.code).to.equal(200); });", + "var jsonData = pm.response.json();", + "var vk = (jsonData && (jsonData.virtual_key || jsonData)) || null;", + "pm.test('expires_at moved out to about 2h from now', function() {", + " pm.expect(vk).to.be.an('object');", + " pm.expect(vk.expires_at).to.be.a('string').and.not.be.empty;", + " var exp = new Date(vk.expires_at).getTime();", + " pm.expect(exp).to.be.greaterThan(Date.now() + 110 * 60000);", + " pm.expect(exp).to.be.lessThan(Date.now() + 130 * 60000);", + "});" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "Set expiry with timezone offset", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "// now+1h expressed in a +05:30 offset instead of Z; the backend must", + "// store the same instant.", + "var target = Date.now() + 3600000;", + "var shifted = new Date(target + 330 * 60000);", + "var offsetTimestamp = shifted.toISOString().slice(0, 19) + '+05:30';", + "pm.request.body.raw = JSON.stringify({expires_at: offsetTimestamp});" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Offset timestamp is accepted', function() { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "GET normalizes offset expiry to the same instant", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('GET VK returns 200', function() { pm.expect(pm.response.code).to.equal(200); });", + "var jsonData = pm.response.json();", + "var vk = (jsonData && (jsonData.virtual_key || jsonData)) || null;", + "pm.test('Stored expiry equals the +05:30 instant (about 1h from now)', function() {", + " pm.expect(vk).to.be.an('object');", + " pm.expect(vk.expires_at).to.be.a('string').and.not.be.empty;", + " var exp = new Date(vk.expires_at).getTime();", + " pm.expect(exp).to.be.greaterThan(Date.now() + 50 * 60000);", + " pm.expect(exp).to.be.lessThan(Date.now() + 70 * 60000);", + "});" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "Update with invalid timestamp is rejected", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Invalid timestamp returns 400', function() { pm.expect(pm.response.code).to.equal(400); });", + "pm.test('Error mentions RFC3339', function() { pm.expect(pm.response.text()).to.include('RFC3339'); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\"expires_at\": \"not-a-timestamp\"}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "Update with past expiry is rejected", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "var pastExpiry = new Date(Date.now() - 60000).toISOString();", + "pm.request.body.raw = JSON.stringify({expires_at: pastExpiry});" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Past expiry returns 400', function() { pm.expect(pm.response.code).to.equal(400); });", + "pm.test('Error mentions future timestamp', function() { pm.expect(pm.response.text()).to.include('future'); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "Clear expiry with empty string", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Clear expiry returns 2xx', function() { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\"expires_at\": \"\"}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "GET shows no expiry after clear", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('GET VK returns 200', function() { pm.expect(pm.response.code).to.equal(200); });", + "var jsonData = pm.response.json();", + "var vk = (jsonData && (jsonData.virtual_key || jsonData)) || null;", + "pm.test('expires_at is cleared', function() {", + " pm.expect(vk).to.be.an('object');", + " pm.expect(vk.expires_at == null).to.be.true;", + "});" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + } + ] + }, + { + "name": "Enforcement", + "item": [ + { + "name": "Re-set future expiry", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "var futureExpiry = new Date(Date.now() + 3600000).toISOString();", + "pm.request.body.raw = JSON.stringify({expires_at: futureExpiry});" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Set expiry returns 2xx', function() { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "Inference with unexpired VK is not blocked as expired", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// The provider call itself may fail in some environments; this test only", + "// guards that governance does not reject an unexpired VK.", + "pm.test('Response carries no expiry rejection', function() {", + " pm.expect(pm.response.text()).to.not.include('Virtual key has expired');", + " pm.expect(pm.response.text()).to.not.include('virtual_key_blocked');", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-bf-vk", + "value": "{{vk_value}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\"model\": \"{{chat_model}}\", \"messages\": [{\"role\": \"user\", \"content\": \"ok\"}], \"max_tokens\": 5}" + }, + "url": { + "raw": "{{base_url}}/v1/chat/completions", + "host": [ + "{{base_url}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "MCP tool execution with unexpired VK is not blocked", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (!pm.collectionVariables.get('mcp_tool_name')) { pm.test('Skipped: no connected MCP client with tools', function() { pm.expect(true).to.be.true; }); } else {", + " pm.test('Unexpired VK MCP execution carries no governance rejection', function() {", + " pm.expect(pm.response.text(), 'body: ' + pm.response.text()).to.not.include('virtual_key_blocked');", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-bf-vk", + "value": "{{vk_value}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\"id\": \"call_expiry_test\", \"type\": \"function\", \"function\": {\"name\": \"{{mcp_tool_name}}\", \"arguments\": \"{}\"}}" + }, + "url": { + "raw": "{{base_url}}/v1/mcp/tool/execute", + "host": [ + "{{base_url}}" + ], + "path": [ + "v1", + "mcp", + "tool", + "execute" + ] + } + } + }, + { + "name": "Set short expiry", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "// Record the absolute expiry instant so the wait before the blocked-request", + "// checks can anchor to it instead of guessing a fixed window.", + "var shortExpiryEpoch = Date.now() + 5000;", + "pm.collectionVariables.set('short_expiry_epoch', String(shortExpiryEpoch));", + "pm.request.body.raw = JSON.stringify({expires_at: new Date(shortExpiryEpoch).toISOString()});" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Set short expiry returns 2xx', function() { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "Inference after expiry is blocked", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "// Wait until the wall clock is safely past the expiry instant recorded by", + "// 'Set short expiry'. Anchoring to the absolute epoch means PUT round-trip", + "// time or a slow runner can never shrink the wait below the expiry boundary.", + "// Newman's sandbox waits for pending timers before sending the request.", + "var deadline = Number(pm.collectionVariables.get('short_expiry_epoch') || 0) + 2000;", + "(function waitForExpiry() {", + " if (Date.now() < deadline) {", + " setTimeout(waitForExpiry, 250);", + " }", + "})();" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Expired VK inference returns 403', function() {", + " pm.expect(pm.response.code, 'body: ' + pm.response.text()).to.equal(403);", + "});", + "pm.test('Rejection is virtual_key_blocked with expired reason', function() {", + " pm.expect(pm.response.text()).to.include('virtual_key_blocked');", + " pm.expect(pm.response.text().toLowerCase()).to.include('expired');", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-bf-vk", + "value": "{{vk_value}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\"model\": \"{{chat_model}}\", \"messages\": [{\"role\": \"user\", \"content\": \"ok\"}], \"max_tokens\": 5}" + }, + "url": { + "raw": "{{base_url}}/v1/chat/completions", + "host": [ + "{{base_url}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "MCP tool execution with expired VK is blocked", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (!pm.collectionVariables.get('mcp_tool_name')) { pm.test('Skipped: no connected MCP client with tools', function() { pm.expect(true).to.be.true; }); } else {", + " pm.test('Expired VK MCP execution returns 403', function() {", + " pm.expect(pm.response.code, 'body: ' + pm.response.text()).to.equal(403);", + " });", + " pm.test('Rejection is virtual_key_blocked with expired reason', function() {", + " pm.expect(pm.response.text()).to.include('virtual_key_blocked');", + " pm.expect(pm.response.text().toLowerCase()).to.include('expired');", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-bf-vk", + "value": "{{vk_value}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\"id\": \"call_expiry_test\", \"type\": \"function\", \"function\": {\"name\": \"{{mcp_tool_name}}\", \"arguments\": \"{}\"}}" + }, + "url": { + "raw": "{{base_url}}/v1/mcp/tool/execute", + "host": [ + "{{base_url}}" + ], + "path": [ + "v1", + "mcp", + "tool", + "execute" + ] + } + } + }, + { + "name": "Expired VK stays visible via GET", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('GET expired VK returns 200', function() { pm.expect(pm.response.code).to.equal(200); });", + "var jsonData = pm.response.json();", + "var vk = (jsonData && (jsonData.virtual_key || jsonData)) || null;", + "pm.test('Expired VK is still active with a past expires_at', function() {", + " pm.expect(vk).to.be.an('object');", + " pm.expect(vk.is_active).to.not.equal(false);", + " pm.expect(vk.expires_at).to.be.a('string').and.not.be.empty;", + " pm.expect(new Date(vk.expires_at).getTime()).to.be.lessThan(Date.now());", + "});" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "Editing an expired VK without touching expiry succeeds", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// The stored expiry is already in the past; an update that omits expires_at", + "// must not re-validate it (omit means leave unchanged).", + "pm.test('Unrelated edit of expired VK returns 2xx', function() { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\"description\": \"edited after expiry\"}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "Deactivate expired VK", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Deactivate returns 2xx', function() { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\"is_active\": false}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "Inactive wins before expired", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Inactive+expired VK inference returns 403', function() {", + " pm.expect(pm.response.code, 'body: ' + pm.response.text()).to.equal(403);", + "});", + "pm.test('Rejection reason is inactive, not expired', function() {", + " pm.expect(pm.response.text().toLowerCase()).to.include('inactive');", + " pm.expect(pm.response.text().toLowerCase()).to.not.include('expired');", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-bf-vk", + "value": "{{vk_value}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\"model\": \"{{chat_model}}\", \"messages\": [{\"role\": \"user\", \"content\": \"ok\"}], \"max_tokens\": 5}" + }, + "url": { + "raw": "{{base_url}}/v1/chat/completions", + "host": [ + "{{base_url}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "MCP tool execution with inactive VK is blocked as inactive", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (!pm.collectionVariables.get('mcp_tool_name')) { pm.test('Skipped: no connected MCP client with tools', function() { pm.expect(true).to.be.true; }); } else {", + " pm.test('Inactive VK MCP execution returns 403', function() {", + " pm.expect(pm.response.code, 'body: ' + pm.response.text()).to.equal(403);", + " });", + " pm.test('Rejection reason is inactive, not expired', function() {", + " pm.expect(pm.response.text().toLowerCase()).to.include('inactive');", + " pm.expect(pm.response.text().toLowerCase()).to.not.include('expired');", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-bf-vk", + "value": "{{vk_value}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\"id\": \"call_expiry_test\", \"type\": \"function\", \"function\": {\"name\": \"{{mcp_tool_name}}\", \"arguments\": \"{}\"}}" + }, + "url": { + "raw": "{{base_url}}/v1/mcp/tool/execute", + "host": [ + "{{base_url}}" + ], + "path": [ + "v1", + "mcp", + "tool", + "execute" + ] + } + } + }, + { + "name": "Reactivate expired VK (expiry untouched)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Reactivate returns 2xx', function() { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\"is_active\": true}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "Reactivated but still expired VK stays blocked", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Reactivation must not implicitly un-expire the key.", + "pm.test('Still returns 403 after reactivation', function() {", + " pm.expect(pm.response.code, 'body: ' + pm.response.text()).to.equal(403);", + "});", + "pm.test('Rejection is still the expired reason', function() {", + " pm.expect(pm.response.text()).to.include('virtual_key_blocked');", + " pm.expect(pm.response.text().toLowerCase()).to.include('expired');", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-bf-vk", + "value": "{{vk_value}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\"model\": \"{{chat_model}}\", \"messages\": [{\"role\": \"user\", \"content\": \"ok\"}], \"max_tokens\": 5}" + }, + "url": { + "raw": "{{base_url}}/v1/chat/completions", + "host": [ + "{{base_url}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Extend expiry of expired VK", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "var extended = new Date(Date.now() + 3600000).toISOString();", + "pm.request.body.raw = JSON.stringify({expires_at: extended});" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Extend expired VK returns 2xx', function() { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "GET reflects extension into the future", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('GET VK returns 200', function() { pm.expect(pm.response.code).to.equal(200); });", + "var jsonData = pm.response.json();", + "var vk = (jsonData && (jsonData.virtual_key || jsonData)) || null;", + "pm.test('expires_at is about 1h from now', function() {", + " pm.expect(vk).to.be.an('object');", + " pm.expect(vk.expires_at).to.be.a('string').and.not.be.empty;", + " var exp = new Date(vk.expires_at).getTime();", + " pm.expect(exp).to.be.greaterThan(Date.now() + 50 * 60000);", + " pm.expect(exp).to.be.lessThan(Date.now() + 70 * 60000);", + "});" + ] + } + } + ], + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "Inference works after extending expiry", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Extension restores access (no expiry rejection)', function() {", + " pm.expect(pm.response.text()).to.not.include('virtual_key_blocked');", + " pm.expect(pm.response.text()).to.not.include('Virtual key has expired');", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-bf-vk", + "value": "{{vk_value}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\"model\": \"{{chat_model}}\", \"messages\": [{\"role\": \"user\", \"content\": \"ok\"}], \"max_tokens\": 5}" + }, + "url": { + "raw": "{{base_url}}/v1/chat/completions", + "host": [ + "{{base_url}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Clear expiry", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Clear expiry returns 2xx', function() { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "PUT", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\"expires_at\": \"\"}" + }, + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "Inference after clearing expiry is not blocked", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Response carries no governance rejection', function() {", + " pm.expect(pm.response.text()).to.not.include('virtual_key_blocked');", + " pm.expect(pm.response.text()).to.not.include('Virtual key has expired');", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-bf-vk", + "value": "{{vk_value}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\"model\": \"{{chat_model}}\", \"messages\": [{\"role\": \"user\", \"content\": \"ok\"}], \"max_tokens\": 5}" + }, + "url": { + "raw": "{{base_url}}/v1/chat/completions", + "host": [ + "{{base_url}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Teardown", + "item": [ + { + "name": "Delete VK", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Delete VK returns 2xx', function() { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ], + "request": { + "method": "DELETE", + "header": [], + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk_id}}" + ] + } + } + }, + { + "name": "Delete future-expiry VK", + "event": [ + { + "listen": "prerequest", + "script": { + "type": "text/javascript", + "exec": [ + "// Nothing to delete when the future-expiry VK was never created; skip so the", + "// request doesn't fire a DELETE against the collection route with no ID.", + "if (!pm.collectionVariables.get('vk2_id') && pm.execution && typeof pm.execution.skipRequest === 'function') {", + " pm.execution.skipRequest();", + "}" + ] + } + }, + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.collectionVariables.get('vk2_id')) {", + " pm.test('Delete future-expiry VK returns 2xx', function() { pm.expect(pm.response.code).to.be.within(200, 299); });", + "} else {", + " pm.test('Skipped: future-expiry VK was never created', function() { pm.expect(true).to.be.true; });", + "}" + ] + } + } + ], + "request": { + "method": "DELETE", + "header": [], + "url": { + "raw": "{{base_url}}/api/governance/virtual-keys/{{vk2_id}}", + "host": [ + "{{base_url}}" + ], + "path": [ + "api", + "governance", + "virtual-keys", + "{{vk2_id}}" + ] + } + } + } + ] + } + ] +} diff --git a/tests/e2e/api/collections/provider-harness.json b/tests/e2e/api/collections/provider-harness.json index 0389efe000b..297a67a56a0 100644 --- a/tests/e2e/api/collections/provider-harness.json +++ b/tests/e2e/api/collections/provider-harness.json @@ -126,6 +126,12 @@ " } else if (j.object === 'file' && typeof j.id === 'string') {", " shape = 'file-object';", " hasContent = !!j.id;", + " } else if (j.type === 'file' && typeof j.id === 'string') {", + " shape = 'anthropic-file-object';", + " hasContent = !!j.id;", + " } else if (j.type === 'file_deleted' && typeof j.id === 'string') {", + " shape = 'anthropic-file-deleted';", + " hasContent = !!j.id;", " }", " pm.expect(hasContent, 'expected non-empty content (shape=' + shape + ', body=' + JSON.stringify(j).slice(0, 200) + ')').to.be.true;", " });", @@ -135,21 +141,81 @@ } ], "variable": [ - { "key": "baseUrl", "value": "http://localhost:8080", "type": "string" }, - { "key": "openaiKey", "value": "sk-replace-me", "type": "string" }, - { "key": "anthropicKey", "value": "sk-ant-replace-me", "type": "string" }, - { "key": "genaiKey", "value": "AIza-replace-me", "type": "string" }, - { "key": "bedrockModel", "value": "global.anthropic.claude-opus-4-7", "type": "string" }, - { "key": "azureDeployment", "value": "gpt-4o", "type": "string" }, - { "key": "azureApiVersion", "value": "2024-10-21", "type": "string" }, - { "key": "genaiModel", "value": "gemini-2.5-pro", "type": "string" }, - { "key": "vertexModel", "value": "gemini-2.5-pro", "type": "string" }, - { "key": "bedrockGuardrailIdentifier", "value": "", "type": "string" }, - { "key": "bedrockGuardrailVersion", "value": "DRAFT", "type": "string" }, - { "key": "vertexGcsBucket", "value": "replace-me-bucket", "type": "string" }, - { "key": "vertexGcsPrefix", "value": "bifrost-e2e/", "type": "string" }, - { "key": "vertexProject", "value": "replace-me-project", "type": "string" }, - { "key": "vertexLocation", "value": "us-central1", "type": "string" } + { + "key": "baseUrl", + "value": "http://localhost:8080", + "type": "string" + }, + { + "key": "openaiKey", + "value": "sk-replace-me", + "type": "string" + }, + { + "key": "anthropicKey", + "value": "sk-ant-replace-me", + "type": "string" + }, + { + "key": "genaiKey", + "value": "AIza-replace-me", + "type": "string" + }, + { + "key": "bedrockModel", + "value": "global.anthropic.claude-opus-4-7", + "type": "string" + }, + { + "key": "azureDeployment", + "value": "gpt-4o", + "type": "string" + }, + { + "key": "azureApiVersion", + "value": "2024-10-21", + "type": "string" + }, + { + "key": "genaiModel", + "value": "gemini-2.5-pro", + "type": "string" + }, + { + "key": "vertexModel", + "value": "gemini-2.5-pro", + "type": "string" + }, + { + "key": "bedrockGuardrailIdentifier", + "value": "", + "type": "string" + }, + { + "key": "bedrockGuardrailVersion", + "value": "DRAFT", + "type": "string" + }, + { + "key": "vertexGcsBucket", + "value": "replace-me-bucket", + "type": "string" + }, + { + "key": "vertexGcsPrefix", + "value": "bifrost-e2e/", + "type": "string" + }, + { + "key": "vertexProject", + "value": "replace-me-project", + "type": "string" + }, + { + "key": "vertexLocation", + "value": "us-central1", + "type": "string" + } ], "item": [ { @@ -161,7 +227,10 @@ "request": { "method": "POST", "header": [ - { "key": "Content-Type", "value": "application/json" } + { + "key": "Content-Type", + "value": "application/json" + } ], "body": { "mode": "raw", @@ -169,8 +238,14 @@ }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", - "host": ["{{baseUrl}}"], - "path": ["v1", "chat", "completions"] + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] } } }, @@ -179,7 +254,10 @@ "request": { "method": "POST", "header": [ - { "key": "Content-Type", "value": "application/json" } + { + "key": "Content-Type", + "value": "application/json" + } ], "body": { "mode": "raw", @@ -187,8 +265,13 @@ }, "url": { "raw": "{{baseUrl}}/v1/responses", - "host": ["{{baseUrl}}"], - "path": ["v1", "responses"] + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] } } } @@ -203,8 +286,14 @@ "request": { "method": "POST", "header": [ - { "key": "Content-Type", "value": "application/json" }, - { "key": "Authorization", "value": "Bearer {{openaiKey}}" } + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } ], "body": { "mode": "raw", @@ -212,8 +301,15 @@ }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", - "host": ["{{baseUrl}}"], - "path": ["openai", "v1", "chat", "completions"] + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] } } }, @@ -222,8 +318,14 @@ "request": { "method": "POST", "header": [ - { "key": "Content-Type", "value": "application/json" }, - { "key": "Authorization", "value": "Bearer {{openaiKey}}" } + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } ], "body": { "mode": "raw", @@ -231,8 +333,14 @@ }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", - "host": ["{{baseUrl}}"], - "path": ["openai", "v1", "responses"] + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] } } } @@ -247,9 +355,18 @@ "request": { "method": "POST", "header": [ - { "key": "Content-Type", "value": "application/json" }, - { "key": "x-api-key", "value": "{{anthropicKey}}" }, - { "key": "anthropic-version", "value": "2023-06-01" } + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } ], "body": { "mode": "raw", @@ -257,8 +374,14 @@ }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", - "host": ["{{baseUrl}}"], - "path": ["anthropic", "v1", "messages"] + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] } } } @@ -273,7 +396,10 @@ "request": { "method": "POST", "header": [ - { "key": "Content-Type", "value": "application/json" } + { + "key": "Content-Type", + "value": "application/json" + } ], "body": { "mode": "raw", @@ -281,8 +407,15 @@ }, "url": { "raw": "{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse", - "host": ["{{baseUrl}}"], - "path": ["bedrock", "model", "{{bedrockModel}}", "converse"] + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "{{bedrockModel}}", + "converse" + ] } } }, @@ -291,7 +424,10 @@ "request": { "method": "POST", "header": [ - { "key": "Content-Type", "value": "application/json" } + { + "key": "Content-Type", + "value": "application/json" + } ], "body": { "mode": "raw", @@ -299,8 +435,15 @@ }, "url": { "raw": "{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse", - "host": ["{{baseUrl}}"], - "path": ["bedrock", "model", "{{bedrockModel}}", "converse"] + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "{{bedrockModel}}", + "converse" + ] } } }, @@ -332,7 +475,10 @@ "request": { "method": "POST", "header": [ - { "key": "Content-Type", "value": "application/json" } + { + "key": "Content-Type", + "value": "application/json" + } ], "body": { "mode": "raw", @@ -340,8 +486,15 @@ }, "url": { "raw": "{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse", - "host": ["{{baseUrl}}"], - "path": ["bedrock", "model", "{{bedrockModel}}", "converse"] + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "{{bedrockModel}}", + "converse" + ] } } }, @@ -366,6 +519,11 @@ " var deltas = (body.match(/contentBlockDelta/g) || []).length;", " pm.expect(deltas, 'expected >= 2 contentBlockDelta frames, got ' + deltas).to.be.at.least(2);", " pm.expect(body, 'expected a messageStop event terminating the stream').to.include('messageStop');", + "});", + "pm.test('Bedrock converse-stream closes content blocks (contentBlockStop before messageStop) - #4923', function () {", + " var b2 = pm.response.text() || '';", + " pm.expect(b2, 'expected a contentBlockStop event closing each content block').to.include('contentBlockStop');", + " pm.expect(b2.indexOf('contentBlockStop'), 'contentBlockStop must precede the terminal messageStop').to.be.below(b2.lastIndexOf('messageStop'));", "});" ] } @@ -457,6 +615,67 @@ ] } } + }, + { + "name": "POST /v1/chat/completions — bedrock streaming (SSE incremental deltas)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Bedrock streaming via the OpenAI-compatible SSE path (/v1/chat/completions).", + "// Confirms chunked SSE delivery with multiple incremental content deltas and a", + "// [DONE] terminator. Note: newman buffers the full response, so this validates", + "// SSE framing/chunking, not wall-clock TTFB (that is covered by the Go unit test", + "// TestChatCompletionStream_StreamsIncrementally_NotBuffered). Issue #4542.", + "if (pm.response.code >= 400) { return; }", + "pm.test('Bedrock SSE: content-type is text/event-stream', function () {", + " var ct = pm.response.headers.get('content-type') || '';", + " pm.expect(ct, 'expected SSE, got ' + ct).to.include('text/event-stream');", + "});", + "pm.test('Bedrock SSE: multiple data chunks with incremental content deltas', function () {", + " var lines = (pm.response.text() || '').split('\\n');", + " var chunks = 0, sawDelta = false, sawDone = false;", + " lines.forEach(function (l) {", + " if (l.indexOf('data: ') !== 0) { return; }", + " chunks++;", + " var p = l.slice(6).trim();", + " if (p === '[DONE]') { sawDone = true; return; }", + " try { var j = JSON.parse(p); var d = j.choices && j.choices[0] && j.choices[0].delta; if (d && typeof d.content === 'string' && d.content.length) { sawDelta = true; } } catch (e) {}", + " });", + " pm.expect(chunks, 'expected > 1 SSE data chunks, got ' + chunks).to.be.above(1);", + " pm.expect(sawDelta, 'expected at least one chunk with an incremental content delta').to.be.true;", + " pm.expect(sawDone, 'expected an SSE [DONE] terminator').to.be.true;", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"stream\": true,\n \"max_tokens\": 256,\n \"messages\": [\n { \"role\": \"user\", \"content\": \"Write three short sentences about the Unix philosophy.\" }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } } ] }, @@ -469,8 +688,14 @@ "request": { "method": "POST", "header": [ - { "key": "Content-Type", "value": "application/json" }, - { "key": "api-key", "value": "{{openaiKey}}" } + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } ], "body": { "mode": "raw", @@ -478,10 +703,22 @@ }, "url": { "raw": "{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", - "host": ["{{baseUrl}}"], - "path": ["openai", "openai", "deployments", "{{azureDeployment}}", "chat", "completions"], + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], "query": [ - { "key": "api-version", "value": "{{azureApiVersion}}" } + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } ] } } @@ -497,8 +734,14 @@ "request": { "method": "POST", "header": [ - { "key": "Content-Type", "value": "application/json" }, - { "key": "x-goog-api-key", "value": "{{genaiKey}}" } + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } ], "body": { "mode": "raw", @@ -506,8 +749,15 @@ }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", - "host": ["{{baseUrl}}"], - "path": ["genai", "v1beta", "models", "{{genaiModel}}:generateContent"] + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] } } } @@ -522,8 +772,14 @@ "request": { "method": "POST", "header": [ - { "key": "Content-Type", "value": "application/json" }, - { "key": "x-goog-api-key", "value": "{{genaiKey}}" } + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } ], "body": { "mode": "raw", @@ -531,8 +787,15 @@ }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent", - "host": ["{{baseUrl}}"], - "path": ["genai", "v1beta", "models", "{{vertexModel}}:generateContent"] + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:generateContent" + ] } } } @@ -550,1080 +813,28739 @@ "name": "8.1.A Native /v1/chat/completions (unified, 10 per provider)", "description": "The headline demo: one endpoint, one request shape, only the `model` field varies. 5 providers x 10 models = 50 requests (Azure intentionally skipped — Azure deployments use a different URL shape, covered separately in 8.1.C and the legacy §5).", "item": [ - { - "name": "OpenAI (10 latest)", - "item": [ - { "name": "openai/gpt-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "openai/gpt-5-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-5-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "openai/gpt-5-nano", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-5-nano\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "openai/gpt-4.1", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-4.1\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "openai/gpt-4.1-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-4.1-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "openai/gpt-4.1-nano", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-4.1-nano\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "openai/gpt-4o", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "openai/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "openai/o3", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/o3\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "openai/o3-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/o3-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } } - ] - }, - { - "name": "Anthropic (10 latest)", - "item": [ - { "name": "anthropic/claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "anthropic/claude-sonnet-4-6", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "anthropic/claude-haiku-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "anthropic/claude-opus-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-opus-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "anthropic/claude-sonnet-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "anthropic/claude-opus-4-6", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-opus-4-6\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "anthropic/claude-opus-4-8", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "anthropic/claude-sonnet-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "anthropic/claude-opus-4-1", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-opus-4-1\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "anthropic/claude-haiku-4-5-20251001", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5-20251001\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } } - ] - }, - { - "name": "Bedrock (10 latest)", - "item": [ - { "name": "bedrock/global.anthropic.claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "bedrock/global.anthropic.claude-sonnet-4-6", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "bedrock/global.anthropic.claude-opus-4-5-20251101-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-5-20251101-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "bedrock/global.anthropic.claude-sonnet-4-5-20250929-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-5-20250929-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "bedrock/us.amazon.nova-pro-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.amazon.nova-pro-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "bedrock/us.amazon.nova-lite-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "bedrock/us.amazon.nova-micro-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.amazon.nova-micro-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "bedrock/us.meta.llama3-1-70b-instruct-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.meta.llama3-1-70b-instruct-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "bedrock/global.anthropic.claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } } - ] - }, - { - "name": "Gemini / GenAI (10 latest)", - "item": [ - { "name": "gemini/gemini-2.5-pro", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "gemini/gemini-2.5-flash-lite", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash-lite\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "gemini/gemini-3.1-pro-preview", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-3.1-pro-preview\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "gemini/gemini-2.5-flash-lite", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash-lite\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "gemini/gemini-3-flash-preview", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-3-flash-preview\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "gemini/gemini-3.1-flash-lite-preview", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-3.1-flash-lite-preview\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "gemini/gemini-flash-latest", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-flash-latest\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "gemini/gemini-pro-latest", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-pro-latest\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } } - ] - }, - { - "name": "Vertex (10 latest)", - "item": [ - { "name": "vertex/gemini-2.5-pro", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "vertex/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "vertex/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "vertex/claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "vertex/gemini-2.5-flash-lite", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-2.5-flash-lite\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "[PREVIEW] vertex/gemini-3-flash-preview", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-3-flash-preview\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "[PREVIEW] vertex/gemini-3.1-pro-preview", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-3.1-pro-preview\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "[PREVIEW] vertex/gemini-pro-latest", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-pro-latest\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "[PREVIEW] vertex/gemini-flash-latest", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-flash-latest\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "[PREVIEW] vertex/gemini-3.1-flash-lite-preview", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-3.1-flash-lite-preview\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } } - ] - } - ] - }, - { - "name": "8.1.B Native /v1/responses (Responses shape × all providers)", - "description": "OpenAI Responses-API request shape (`input` instead of `messages`) against the native endpoint with model strings from all 6 providers. Bifrost's Responses-shape converter normalizes to BifrostChatRequest internally; the same converter must handle every backend.", - "item": [ - { "name": "openai/gpt-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-5\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "openai/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "anthropic/claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "anthropic/claude-haiku-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "gemini/gemini-2.5-pro", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "vertex/gemini-2.5-pro", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "vertex/claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "bedrock/global.anthropic.claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "bedrock/us.amazon.nova-lite-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "azure/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "azure/gpt-4o", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-4o\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "xai/grok-4-0709", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } } - ] - }, - { - "name": "8.1.C OpenAI drop-in /openai/v1/chat/completions × non-OpenAI providers", - "description": "OpenAI ChatCompletion shape via the `/openai` drop-in prefix, routed to non-OpenAI backends via `provider/` model prefix. The OpenAI diagonal is already in §2 and 8.1.A; here we exercise the drop-in's converter against backends that need shape translation.", - "item": [ - { "name": "anthropic/claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "chat", "completions"] } } }, - { "name": "anthropic/claude-haiku-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "chat", "completions"] } } }, - { "name": "gemini/gemini-2.5-pro", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "chat", "completions"] } } }, - { "name": "gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "chat", "completions"] } } }, - { "name": "vertex/gemini-2.5-pro", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "chat", "completions"] } } }, - { "name": "vertex/claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "chat", "completions"] } } }, - { "name": "bedrock/global.anthropic.claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "chat", "completions"] } } }, - { "name": "bedrock/us.amazon.nova-lite-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "chat", "completions"] } } }, - { "name": "azure/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "chat", "completions"] } } }, - { "name": "azure/gpt-4o", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-4o\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "chat", "completions"] } } } - ] - }, - { - "name": "8.1.D OpenAI drop-in /openai/v1/responses × non-OpenAI providers", - "description": "OpenAI Responses-API shape via `/openai` drop-in, routed to non-OpenAI backends. Responses-shape converter must round-trip through every provider's chat backend.", - "item": [ - { "name": "anthropic/claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "responses"] } } }, - { "name": "anthropic/claude-haiku-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "responses"] } } }, - { "name": "gemini/gemini-2.5-pro", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "responses"] } } }, - { "name": "gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "responses"] } } }, - { "name": "vertex/gemini-2.5-pro", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "responses"] } } }, - { "name": "vertex/claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "responses"] } } }, - { "name": "bedrock/global.anthropic.claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "responses"] } } }, - { "name": "bedrock/us.amazon.nova-lite-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "responses"] } } }, - { "name": "azure/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "responses"] } } }, - { "name": "azure/gpt-4o", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-4o\",\n \"input\": \"Hello\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "responses"] } } } - ] - }, - { - "name": "8.1.E Anthropic drop-in /anthropic/v1/messages × non-Anthropic providers", - "description": "Anthropic Messages-API shape (requires `max_tokens`) routed to non-Anthropic backends. Tests Bifrost's Messages→BifrostChatRequest converter against every other provider.", - "item": [ - { "name": "openai/gpt-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-5\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "openai/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "gemini/gemini-2.5-pro", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "vertex/gemini-2.5-pro", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "vertex/claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "bedrock/global.anthropic.claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "bedrock/us.amazon.nova-lite-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "azure/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "azure/gpt-4o", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-4o\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } } - ] - }, - { - "name": "8.1.F GenAI drop-in /genai/v1beta/models/{model}:generateContent × non-Gemini providers", - "description": "Google GenAI generateContent shape (`contents` / `parts` body, model in URL path) routed to non-Gemini backends. The provider/model prefix is part of the path; Bifrost's GenAI router (`integrations/router.go`) parses it before invoking the backend.", - "item": [ - { "name": "openai/gpt-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/openai/gpt-5:generateContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "openai", "gpt-5:generateContent"] } } }, - { "name": "openai/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/openai/gpt-4o-mini:generateContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "openai", "gpt-4o-mini:generateContent"] } } }, - { "name": "anthropic/claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/anthropic/claude-opus-4-7:generateContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "anthropic", "claude-opus-4-7:generateContent"] } } }, - { "name": "anthropic/claude-haiku-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/anthropic/claude-haiku-4-5:generateContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "anthropic", "claude-haiku-4-5:generateContent"] } } }, - { "name": "vertex/gemini-2.5-pro", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/vertex/gemini-2.5-pro:generateContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "vertex", "gemini-2.5-pro:generateContent"] } } }, - { "name": "vertex/claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/vertex/claude-opus-4-7:generateContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "vertex", "claude-opus-4-7:generateContent"] } } }, - { "name": "bedrock/global.anthropic.claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/bedrock/global.anthropic.claude-opus-4-7:generateContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "bedrock", "global.anthropic.claude-opus-4-7:generateContent"] } } }, - { "name": "bedrock/us.amazon.nova-lite-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/bedrock/us.amazon.nova-lite-v1:0:generateContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "bedrock", "us.amazon.nova-lite-v1:0:generateContent"] } } }, - { "name": "azure/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/azure/gpt-4o-mini:generateContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "azure", "gpt-4o-mini:generateContent"] } } }, - { "name": "azure/gpt-4o", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/azure/gpt-4o:generateContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "azure", "gpt-4o:generateContent"] } } } - ] - } - ] - }, - { - "name": "8.2 Text Chat (streaming)", - "description": "Streaming variants of the 8.1 matrix. OpenAI / Anthropic / Responses shapes use `stream: true` in the body; GenAI uses `:streamGenerateContent?alt=sse`; Bedrock uses the `/converse-stream` URL.\n\nThe folder-level test script (inherited by all items) asserts the Content-Type is a streaming type and that the body contains a recognized stream terminator (`[DONE]` for OpenAI SSE, `message_stop` / `message_delta` for Anthropic, `finishReason` for GenAI, `messageStop` for Bedrock).", - "event": [ - { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "// Streaming-specific assertion: header + full body consume + terminator check.", - "// Runs in addition to the collection-level 2xx check. The collection-level content-shape test skips on streaming responses (see line ~34), so this fills the gap.", - "if (pm.response.code >= 400) { return; }", - "pm.test('Streaming response with terminator', function () {", - " var ct = (pm.response.headers.get('content-type') || '').toLowerCase();", - " var isStreamCt = ct.indexOf('event-stream') !== -1 || ct.indexOf('eventstream') !== -1 || ct.indexOf('x-ndjson') !== -1;", - " pm.expect(isStreamCt, 'expected streaming content-type, got: ' + ct).to.be.true;", - " var body = pm.response.text() || '';", - " pm.expect(body.length, 'streaming body was empty').to.be.greaterThan(0);", - " var terms = ['[DONE]', 'message_stop', '\"type\":\"message_delta\"', 'response.completed', 'finishReason', 'messageStop', '\"event\":\"messageStop\"'];", - " var hit = terms.some(function (p) { return body.indexOf(p) !== -1; });", - " pm.expect(hit, 'no recognized stream terminator. First 200 chars: ' + body.slice(0, 200)).to.be.true;", - "});" - ] - } - } - ], - "item": [ - { - "name": "8.2.A Native /v1/chat/completions streaming × all providers", - "description": "Streaming chat completions via the unified native endpoint, one model per provider.", - "item": [ - { "name": "openai/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "anthropic/claude-haiku-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "vertex/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "bedrock/us.amazon.nova-lite-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "azure/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } } - ] - }, - { - "name": "8.2.B Native /v1/responses streaming × all providers", - "description": "Streaming Responses-API via native endpoint.", - "item": [ - { "name": "openai/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "anthropic/claude-haiku-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "vertex/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "bedrock/us.amazon.nova-lite-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "azure/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } }, - { "name": "xai/grok-4-0709", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } } - ] - }, - { - "name": "8.2.C OpenAI drop-in /openai/v1/chat/completions streaming × non-OpenAI providers", - "description": "OpenAI streaming chat shape via drop-in, routed to non-OpenAI backends.", - "item": [ - { "name": "anthropic/claude-haiku-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "chat", "completions"] } } }, - { "name": "gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "chat", "completions"] } } }, - { "name": "vertex/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "chat", "completions"] } } }, - { "name": "bedrock/us.amazon.nova-lite-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "chat", "completions"] } } }, - { "name": "azure/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "chat", "completions"] } } } - ] - }, - { - "name": "8.2.D OpenAI drop-in /openai/v1/responses streaming × non-OpenAI providers", - "description": "OpenAI streaming Responses shape via drop-in, routed to non-OpenAI backends.", - "item": [ - { "name": "anthropic/claude-haiku-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "responses"] } } }, - { "name": "gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "responses"] } } }, - { "name": "vertex/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "responses"] } } }, - { "name": "bedrock/us.amazon.nova-lite-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "responses"] } } }, - { "name": "azure/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/responses", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "responses"] } } } - ] - }, - { - "name": "8.2.E Anthropic drop-in /anthropic/v1/messages streaming × non-Anthropic providers", - "description": "Anthropic streaming Messages-API shape via drop-in, routed to non-Anthropic backends.", - "item": [ - { "name": "openai/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "vertex/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "bedrock/us.amazon.nova-lite-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "azure/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } } - ] - }, - { - "name": "8.2.F GenAI drop-in :streamGenerateContent × non-Gemini providers", - "description": "Google GenAI streaming via `:streamGenerateContent?alt=sse`, routed to non-Gemini backends. Body unchanged from non-streaming generateContent; SSE is requested via the URL suffix + query.", - "item": [ - { "name": "openai/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Count to 5\" }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/openai/gpt-4o-mini:streamGenerateContent?alt=sse", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "openai", "gpt-4o-mini:streamGenerateContent"], "query": [{"key": "alt", "value": "sse"}] } } }, - { "name": "anthropic/claude-haiku-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Count to 5\" }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/anthropic/claude-haiku-4-5:streamGenerateContent?alt=sse", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "anthropic", "claude-haiku-4-5:streamGenerateContent"], "query": [{"key": "alt", "value": "sse"}] } } }, - { "name": "vertex/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Count to 5\" }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/vertex/gemini-2.5-flash:streamGenerateContent?alt=sse", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "vertex", "gemini-2.5-flash:streamGenerateContent"], "query": [{"key": "alt", "value": "sse"}] } } }, - { "name": "bedrock/us.amazon.nova-lite-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Count to 5\" }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/bedrock/us.amazon.nova-lite-v1:0:streamGenerateContent?alt=sse", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "bedrock", "us.amazon.nova-lite-v1:0:streamGenerateContent"], "query": [{"key": "alt", "value": "sse"}] } } }, - { "name": "azure/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Count to 5\" }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/azure/gpt-4o-mini:streamGenerateContent?alt=sse", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "azure", "gpt-4o-mini:streamGenerateContent"], "query": [{"key": "alt", "value": "sse"}] } } } - ] - } - ] - }, - { - "name": "8.3 Embeddings", - "description": "Embedding endpoints × providers. Anthropic returns NewUnsupportedOperationError for Embedding (core/providers/anthropic/anthropic.go:1878-1923) — those cells are tagged `[SKIP]` and skipped unless `include_skip=1`.\n\nResponse shapes (validated by extended top-level detector):\n- OpenAI: `{data: [{embedding: [...]}]}` → list-data branch\n- GenAI: `{embedding: {values: [...]}}` → genai-embedding branch\n- Bedrock Titan: `{embedding: [...]}` → bedrock-titan-embedding branch\n- Vertex: `{predictions: [{embeddings: {values}}]}` → vertex-predictions branch", - "item": [ - { - "name": "8.3.H Native /v1/embeddings × all providers", - "description": "Native embeddings endpoint with model strings from every provider. OpenAI-shape request, but Bifrost routes to the model's actual provider.", - "item": [ - { "name": "openai/text-embedding-3-small", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/text-embedding-3-small\",\n \"input\": \"Hello world\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/embeddings", "host": ["{{baseUrl}}"], "path": ["v1", "embeddings"] } } }, - { "name": "[SKIP] anthropic/claude-haiku-4-5 (no embedding endpoint)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Hello world\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/embeddings", "host": ["{{baseUrl}}"], "path": ["v1", "embeddings"] } } }, - { "name": "gemini/gemini-embedding-001", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-embedding-001\",\n \"input\": \"Hello world\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/embeddings", "host": ["{{baseUrl}}"], "path": ["v1", "embeddings"] } } }, - { "name": "vertex/text-embedding-005", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/text-embedding-005\",\n \"input\": \"Hello world\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/embeddings", "host": ["{{baseUrl}}"], "path": ["v1", "embeddings"] } } }, - { "name": "bedrock/amazon.titan-embed-text-v2:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/amazon.titan-embed-text-v2:0\",\n \"input\": \"Hello world\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/embeddings", "host": ["{{baseUrl}}"], "path": ["v1", "embeddings"] } } }, - { "name": "azure/text-embedding-ada-002", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/text-embedding-ada-002\",\n \"input\": \"Hello world\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/embeddings", "host": ["{{baseUrl}}"], "path": ["v1", "embeddings"] } } } - ] - }, - { - "name": "8.3.I OpenAI drop-in /openai/v1/embeddings × non-OpenAI providers", - "description": "OpenAI Embeddings shape via drop-in, routed to non-OpenAI backends.", - "item": [ - { "name": "[SKIP] anthropic/claude-haiku-4-5 (no embedding endpoint)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Hello world\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/embeddings", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "embeddings"] } } }, - { "name": "gemini/gemini-embedding-001", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-embedding-001\",\n \"input\": \"Hello world\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/embeddings", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "embeddings"] } } }, - { "name": "vertex/text-embedding-005", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/text-embedding-005\",\n \"input\": \"Hello world\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/embeddings", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "embeddings"] } } }, - { "name": "bedrock/amazon.titan-embed-text-v2:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/amazon.titan-embed-text-v2:0\",\n \"input\": \"Hello world\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/embeddings", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "embeddings"] } } }, - { "name": "azure/text-embedding-ada-002", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/text-embedding-ada-002\",\n \"input\": \"Hello world\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/embeddings", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "embeddings"] } } } - ] - }, - { - "name": "8.3.J GenAI drop-in /genai/v1beta/models/{model}:embedContent × non-Gemini providers", - "description": "Google GenAI embedContent shape (`content.parts`), model in URL path. Routed to non-Gemini embedding backends.", - "item": [ - { "name": "openai/text-embedding-3-small", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"content\": { \"parts\": [{ \"text\": \"Hello world\" }] }\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/openai/text-embedding-3-small:embedContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "openai", "text-embedding-3-small:embedContent"] } } }, - { "name": "[SKIP] anthropic/claude-haiku-4-5 (no embedding endpoint)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"content\": { \"parts\": [{ \"text\": \"Hello world\" }] }\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/anthropic/claude-haiku-4-5:embedContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "anthropic", "claude-haiku-4-5:embedContent"] } } }, - { "name": "vertex/text-embedding-005", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"content\": { \"parts\": [{ \"text\": \"Hello world\" }] }\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/vertex/text-embedding-005:embedContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "vertex", "text-embedding-005:embedContent"] } } }, - { "name": "bedrock/amazon.titan-embed-text-v2:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"content\": { \"parts\": [{ \"text\": \"Hello world\" }] }\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/bedrock/amazon.titan-embed-text-v2:0:embedContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "bedrock", "amazon.titan-embed-text-v2:0:embedContent"] } } }, - { "name": "azure/text-embedding-ada-002", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"content\": { \"parts\": [{ \"text\": \"Hello world\" }] }\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/azure/text-embedding-ada-002:embedContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "azure", "text-embedding-ada-002:embedContent"] } } } - ] - } - ] - }, - { - "name": "8.4 Audio (transcription + speech)", - "description": "Audio endpoints × providers. Most providers don't expose a dedicated audio endpoint through Bifrost — those cells are `[SKIP]`:\n- Anthropic: no audio (core/providers/anthropic/anthropic.go:1878-1923)\n- Vertex: no transcription (core/providers/vertex/vertex.go:1722-1729)\n- Bedrock: no transcription (core/providers/bedrock/bedrock.go:1936-1941); audio input rejected (core/providers/bedrock/utils.go:1058)\n- Gemini/Vertex TTS: no /v1/audio/speech equivalent\n\nTranscription tests require `tests/e2e/api/fixtures/sample.mp3` (small audio sample) at runtime — see Phase 7 verification.\n\nResponse shapes (validated by extended top-level detector):\n- Transcription JSON: `{text: \"...\"}` → audio-transcription branch\n- TTS binary: `audio/*` Content-Type → prelude's non-JSON early-return passes status-only check", - "item": [ - { - "name": "8.4.K Native /v1/audio/transcriptions × providers", - "description": "Transcription endpoint with multipart formdata. Supported providers: OpenAI (Whisper, gpt-4o-mini-transcribe), Gemini (via generateContent under the hood), Azure (Whisper deployment).", - "item": [ - { "name": "openai/whisper-1", "request": { "method": "POST", "body": { "mode": "formdata", "formdata": [{ "key": "model", "value": "openai/whisper-1", "type": "text" }, { "key": "file", "src": "tests/e2e/api/fixtures/sample.mp3", "type": "file" }] }, "url": { "raw": "{{baseUrl}}/v1/audio/transcriptions", "host": ["{{baseUrl}}"], "path": ["v1", "audio", "transcriptions"] } } }, - { "name": "[SKIP] anthropic/claude-haiku-4-5 (no transcription endpoint)", "request": { "method": "POST", "body": { "mode": "formdata", "formdata": [{ "key": "model", "value": "anthropic/claude-haiku-4-5", "type": "text" }, { "key": "file", "src": "tests/e2e/api/fixtures/sample.mp3", "type": "file" }] }, "url": { "raw": "{{baseUrl}}/v1/audio/transcriptions", "host": ["{{baseUrl}}"], "path": ["v1", "audio", "transcriptions"] } } }, - { "name": "gemini/gemini-2.5-flash", "request": { "method": "POST", "body": { "mode": "formdata", "formdata": [{ "key": "model", "value": "gemini/gemini-2.5-flash", "type": "text" }, { "key": "file", "src": "tests/e2e/api/fixtures/sample.mp3", "type": "file" }] }, "url": { "raw": "{{baseUrl}}/v1/audio/transcriptions", "host": ["{{baseUrl}}"], "path": ["v1", "audio", "transcriptions"] } } }, - { "name": "[SKIP] vertex/gemini-2.5-flash (no transcription endpoint in Bifrost)", "request": { "method": "POST", "body": { "mode": "formdata", "formdata": [{ "key": "model", "value": "vertex/gemini-2.5-flash", "type": "text" }, { "key": "file", "src": "tests/e2e/api/fixtures/sample.mp3", "type": "file" }] }, "url": { "raw": "{{baseUrl}}/v1/audio/transcriptions", "host": ["{{baseUrl}}"], "path": ["v1", "audio", "transcriptions"] } } }, - { "name": "[SKIP] bedrock/us.amazon.nova-lite-v1:0 (no transcription endpoint in Bifrost)", "request": { "method": "POST", "body": { "mode": "formdata", "formdata": [{ "key": "model", "value": "bedrock/us.amazon.nova-lite-v1:0", "type": "text" }, { "key": "file", "src": "tests/e2e/api/fixtures/sample.mp3", "type": "file" }] }, "url": { "raw": "{{baseUrl}}/v1/audio/transcriptions", "host": ["{{baseUrl}}"], "path": ["v1", "audio", "transcriptions"] } } }, - { "name": "azure/gpt-4o-transcribe", "request": { "method": "POST", "body": { "mode": "formdata", "formdata": [{ "key": "model", "value": "azure/gpt-4o-transcribe", "type": "text" }, { "key": "file", "src": "tests/e2e/api/fixtures/sample.mp3", "type": "file" }] }, "url": { "raw": "{{baseUrl}}/v1/audio/transcriptions", "host": ["{{baseUrl}}"], "path": ["v1", "audio", "transcriptions"] } } } - ] - }, - { - "name": "8.4.L OpenAI drop-in /openai/v1/audio/transcriptions × non-OpenAI providers", - "description": "OpenAI Transcriptions shape via drop-in, routed to non-OpenAI transcription backends.", - "item": [ - { "name": "[SKIP] anthropic/claude-haiku-4-5 (no transcription endpoint)", "request": { "method": "POST", "body": { "mode": "formdata", "formdata": [{ "key": "model", "value": "anthropic/claude-haiku-4-5", "type": "text" }, { "key": "file", "src": "tests/e2e/api/fixtures/sample.mp3", "type": "file" }] }, "url": { "raw": "{{baseUrl}}/openai/v1/audio/transcriptions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "audio", "transcriptions"] } } }, - { "name": "gemini/gemini-2.5-flash", "request": { "method": "POST", "body": { "mode": "formdata", "formdata": [{ "key": "model", "value": "gemini/gemini-2.5-flash", "type": "text" }, { "key": "file", "src": "tests/e2e/api/fixtures/sample.mp3", "type": "file" }] }, "url": { "raw": "{{baseUrl}}/openai/v1/audio/transcriptions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "audio", "transcriptions"] } } }, - { "name": "[SKIP] vertex/gemini-2.5-flash (no transcription endpoint)", "request": { "method": "POST", "body": { "mode": "formdata", "formdata": [{ "key": "model", "value": "vertex/gemini-2.5-flash", "type": "text" }, { "key": "file", "src": "tests/e2e/api/fixtures/sample.mp3", "type": "file" }] }, "url": { "raw": "{{baseUrl}}/openai/v1/audio/transcriptions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "audio", "transcriptions"] } } }, - { "name": "[SKIP] bedrock/us.amazon.nova-lite-v1:0 (no transcription endpoint)", "request": { "method": "POST", "body": { "mode": "formdata", "formdata": [{ "key": "model", "value": "bedrock/us.amazon.nova-lite-v1:0", "type": "text" }, { "key": "file", "src": "tests/e2e/api/fixtures/sample.mp3", "type": "file" }] }, "url": { "raw": "{{baseUrl}}/openai/v1/audio/transcriptions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "audio", "transcriptions"] } } }, - { "name": "azure/gpt-4o-transcribe", "request": { "method": "POST", "body": { "mode": "formdata", "formdata": [{ "key": "model", "value": "azure/gpt-4o-transcribe", "type": "text" }, { "key": "file", "src": "tests/e2e/api/fixtures/sample.mp3", "type": "file" }] }, "url": { "raw": "{{baseUrl}}/openai/v1/audio/transcriptions", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "audio", "transcriptions"] } } } - ] - }, - { - "name": "8.4.M Native /v1/audio/speech × providers", - "description": "Text-to-speech endpoint. Supported: OpenAI, Azure. Anthropic/Gemini/Vertex/Bedrock have no `/v1/audio/speech`-equivalent endpoint in Bifrost.", - "item": [ - { "name": "openai/tts-1", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/tts-1\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/audio/speech", "host": ["{{baseUrl}}"], "path": ["v1", "audio", "speech"] } } }, - { "name": "[SKIP] anthropic/claude-haiku-4-5 (no TTS)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/audio/speech", "host": ["{{baseUrl}}"], "path": ["v1", "audio", "speech"] } } }, - { "name": "[SKIP] gemini/gemini-2.5-flash (no /v1/audio/speech)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/audio/speech", "host": ["{{baseUrl}}"], "path": ["v1", "audio", "speech"] } } }, - { "name": "[SKIP] vertex/gemini-2.5-flash (no /v1/audio/speech)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/audio/speech", "host": ["{{baseUrl}}"], "path": ["v1", "audio", "speech"] } } }, - { "name": "[SKIP] bedrock/us.amazon.nova-lite-v1:0 (no TTS via Bifrost)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/audio/speech", "host": ["{{baseUrl}}"], "path": ["v1", "audio", "speech"] } } }, - { "name": "azure/gpt-4o-mini-tts", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-4o-mini-tts\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/audio/speech", "host": ["{{baseUrl}}"], "path": ["v1", "audio", "speech"] } } } - ] - }, - { - "name": "8.4.N OpenAI drop-in /openai/v1/audio/speech × non-OpenAI providers", - "description": "OpenAI TTS shape via drop-in. Only Azure has a comparable backend.", - "item": [ - { "name": "[SKIP] anthropic/claude-haiku-4-5 (no TTS)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/audio/speech", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "audio", "speech"] } } }, - { "name": "[SKIP] gemini/gemini-2.5-flash (no /v1/audio/speech)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/audio/speech", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "audio", "speech"] } } }, - { "name": "[SKIP] vertex/gemini-2.5-flash (no /v1/audio/speech)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/audio/speech", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "audio", "speech"] } } }, - { "name": "[SKIP] bedrock/us.amazon.nova-lite-v1:0 (no TTS)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/audio/speech", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "audio", "speech"] } } }, - { "name": "azure/gpt-4o-mini-tts", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-4o-mini-tts\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/audio/speech", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "audio", "speech"] } } } - ] - } - ] - }, - { - "name": "8.5 Image generation", - "description": "Image generation endpoints × providers. Anthropic returns NewUnsupportedOperationError for ImageGeneration (core/providers/anthropic/anthropic.go:1878-1923) — those cells are `[SKIP]`.\n\nResponse shapes (validated by extended top-level detector):\n- OpenAI: `{data: [{url|b64_json}]}` → list-data branch\n- Vertex Imagen: `{predictions: [{bytesBase64Encoded}]}` → vertex-predictions branch\n- Bedrock Titan/Nova: `{images: [\"base64\"]}` → bedrock-images branch\n\nOmits /v1/images/edits — needs a binary PNG fixture; will be added later if needed.", - "item": [ - { - "name": "8.5.O Native /v1/images/generations × providers", - "description": "Image generation endpoint with model strings from all providers. OpenAI request shape; Bifrost translates to backend-specific format internally.", - "item": [ - { "name": "openai/gpt-image-1", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-image-1\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/images/generations", "host": ["{{baseUrl}}"], "path": ["v1", "images", "generations"] } } }, - { "name": "[SKIP] anthropic/claude-haiku-4-5 (no image gen)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/images/generations", "host": ["{{baseUrl}}"], "path": ["v1", "images", "generations"] } } }, - { "name": "gemini/imagen-4.0-generate-001", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/imagen-4.0-generate-001\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/images/generations", "host": ["{{baseUrl}}"], "path": ["v1", "images", "generations"] } } }, - { "name": "vertex/imagen-4.0-generate-001", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/imagen-4.0-generate-001\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/images/generations", "host": ["{{baseUrl}}"], "path": ["v1", "images", "generations"] } } }, - { "name": "bedrock/amazon.nova-canvas-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/amazon.nova-canvas-v1:0\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/images/generations", "host": ["{{baseUrl}}"], "path": ["v1", "images", "generations"] } } }, - { "name": "azure/gpt-image-2", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-image-2\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/images/generations", "host": ["{{baseUrl}}"], "path": ["v1", "images", "generations"] } } } - ] - }, - { - "name": "8.5.P OpenAI drop-in /openai/v1/images/generations × non-OpenAI providers", - "description": "OpenAI image-generation shape via drop-in, routed to non-OpenAI backends.", - "item": [ - { "name": "[SKIP] anthropic/claude-haiku-4-5 (no image gen)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/images/generations", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "images", "generations"] } } }, - { "name": "gemini/imagen-4.0-generate-001", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/imagen-4.0-generate-001\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/images/generations", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "images", "generations"] } } }, - { "name": "vertex/imagen-4.0-generate-001", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/imagen-4.0-generate-001\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/images/generations", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "images", "generations"] } } }, - { "name": "bedrock/amazon.nova-canvas-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/amazon.nova-canvas-v1:0\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/images/generations", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "images", "generations"] } } }, - { "name": "azure/gpt-image-2", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"azure/gpt-image-2\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" }, "url": { "raw": "{{baseUrl}}/openai/v1/images/generations", "host": ["{{baseUrl}}"], "path": ["openai", "v1", "images", "generations"] } } } - ] - } - ] - }, - { - "name": "8.6 Feature combinations", - "description": "Feature-specific cross-shape tests: tool calling, vision (image input), JSON / structured output, reasoning / thinking.\n\nEach feature is tested across multiple endpoint shapes with multiple model providers, proving Bifrost's converters preserve feature semantics across shape boundaries (e.g., OpenAI `tools` array → Anthropic `tools` shape when routed to Claude; Anthropic `tools` shape → Gemini `function_declarations` when routed to Gemini).\n\nResponse-shape assertions (`tool_calls` / `tool_use` / `functionCall`) are already in the top-level detector.", - "item": [ - { - "name": "8.6.1 Tool calling × endpoint shapes × providers", - "description": "Function-calling request. Each shape uses its native tool schema (OpenAI `tools[].function`, Anthropic `tools[].input_schema`, GenAI `tools[].function_declarations`, Bedrock `toolConfig.tools[].toolSpec`). The converter must translate to the backend's native tool shape and translate the response back.", - "item": [ - { "name": "8.6.1.A native chat → openai/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"type\": \"function\", \"function\": { \"name\": \"get_weather\", \"description\": \"Get current weather\", \"parameters\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } } }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "8.6.1.A native chat → anthropic/claude-haiku-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"type\": \"function\", \"function\": { \"name\": \"get_weather\", \"description\": \"Get current weather\", \"parameters\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } } }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "8.6.1.A native chat → gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"type\": \"function\", \"function\": { \"name\": \"get_weather\", \"description\": \"Get current weather\", \"parameters\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } } }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "8.6.1.A native chat → bedrock/global.anthropic.claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"type\": \"function\", \"function\": { \"name\": \"get_weather\", \"description\": \"Get current weather\", \"parameters\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } } }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "8.6.1.E /anthropic/v1/messages → openai/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"max_tokens\": 512,\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"name\": \"get_weather\", \"description\": \"Get current weather\", \"input_schema\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } }]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "8.6.1.E /anthropic/v1/messages → gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"max_tokens\": 512,\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"name\": \"get_weather\", \"description\": \"Get current weather\", \"input_schema\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } }]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "8.6.1.E /anthropic/v1/messages → bedrock/us.amazon.nova-lite-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"max_tokens\": 512,\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"name\": \"get_weather\", \"description\": \"Get current weather\", \"input_schema\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } }]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "8.6.1.F /genai → openai/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"What's the weather in SF?\" }] }],\n \"tools\": [{ \"function_declarations\": [{ \"name\": \"get_weather\", \"description\": \"Get current weather\", \"parameters\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/openai/gpt-4o-mini:generateContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "openai", "gpt-4o-mini:generateContent"] } } }, - { "name": "8.6.1.F /genai → anthropic/claude-haiku-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"What's the weather in SF?\" }] }],\n \"tools\": [{ \"function_declarations\": [{ \"name\": \"get_weather\", \"description\": \"Get current weather\", \"parameters\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/anthropic/claude-haiku-4-5:generateContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "anthropic", "claude-haiku-4-5:generateContent"] } } }, - { "name": "8.6.1.F /genai → bedrock/us.amazon.nova-lite-v1:0", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"What's the weather in SF?\" }] }],\n \"tools\": [{ \"function_declarations\": [{ \"name\": \"get_weather\", \"description\": \"Get current weather\", \"parameters\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/bedrock/us.amazon.nova-lite-v1:0:generateContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "bedrock", "us.amazon.nova-lite-v1:0:generateContent"] } } } - ] - }, - { - "name": "8.6.2 Vision (image input) × endpoint shapes × providers", - "description": "Vision request with a 1×1 transparent PNG (base64 inline). Each shape uses its native image-content schema:\n- OpenAI: `content: [{type:\"image_url\", image_url:{url:\"data:image/png;base64,...\"}}]`\n- Anthropic: `content: [{type:\"image\", source:{type:\"base64\", media_type, data}}]`\n- GenAI: `parts: [{inline_data:{mime_type, data}}]`\n- Bedrock: `content: [{image:{format, source:{bytes}}}]`", - "item": [ - { "name": "8.6.2.A native chat → openai/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": [{ \"type\": \"text\", \"text\": \"Describe this image in one word.\" }, { \"type\": \"image_url\", \"image_url\": { \"url\": \"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "8.6.2.A native chat → anthropic/claude-haiku-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": [{ \"type\": \"text\", \"text\": \"Describe this image in one word.\" }, { \"type\": \"image_url\", \"image_url\": { \"url\": \"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "8.6.2.A native chat → gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": [{ \"type\": \"text\", \"text\": \"Describe this image in one word.\" }, { \"type\": \"image_url\", \"image_url\": { \"url\": \"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "8.6.2.A native chat → bedrock/global.anthropic.claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": [{ \"type\": \"text\", \"text\": \"Describe this image in one word.\" }, { \"type\": \"image_url\", \"image_url\": { \"url\": \"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "8.6.2.E /anthropic/v1/messages → openai/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": [{ \"type\": \"text\", \"text\": \"Describe in one word.\" }, { \"type\": \"image\", \"source\": { \"type\": \"base64\", \"media_type\": \"image/png\", \"data\": \"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "8.6.2.E /anthropic/v1/messages → gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": [{ \"type\": \"text\", \"text\": \"Describe in one word.\" }, { \"type\": \"image\", \"source\": { \"type\": \"base64\", \"media_type\": \"image/png\", \"data\": \"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "8.6.2.F /genai → openai/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Describe in one word.\" }, { \"inline_data\": { \"mime_type\": \"image/png\", \"data\": \"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/openai/gpt-4o-mini:generateContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "openai", "gpt-4o-mini:generateContent"] } } }, - { "name": "8.6.2.F /genai → anthropic/claude-haiku-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Describe in one word.\" }, { \"inline_data\": { \"mime_type\": \"image/png\", \"data\": \"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" }, "url": { "raw": "{{baseUrl}}/genai/v1beta/models/anthropic/claude-haiku-4-5:generateContent", "host": ["{{baseUrl}}"], "path": ["genai", "v1beta", "models", "anthropic", "claude-haiku-4-5:generateContent"] } } } - ] - }, - { - "name": "8.6.3 JSON / structured output × providers", - "description": "OpenAI `response_format: {type: \"json_schema\", json_schema: {...}}` via native chat endpoint, routed to each provider's structured-output mechanism (Anthropic tool-use forcing, Gemini responseSchema, Bedrock tool-spec).", - "item": [ - { "name": "openai/gpt-4o-mini", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Reply with a Greeting object whose `text` field is 'hi'.\" }],\n \"response_format\": { \"type\": \"json_schema\", \"json_schema\": { \"name\": \"Greeting\", \"strict\": true, \"schema\": { \"type\": \"object\", \"properties\": { \"text\": { \"type\": \"string\" } }, \"required\": [\"text\"], \"additionalProperties\": false } } }\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "anthropic/claude-haiku-4-5", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Reply with a Greeting object whose `text` field is 'hi'.\" }],\n \"response_format\": { \"type\": \"json_schema\", \"json_schema\": { \"name\": \"Greeting\", \"strict\": true, \"schema\": { \"type\": \"object\", \"properties\": { \"text\": { \"type\": \"string\" } }, \"required\": [\"text\"], \"additionalProperties\": false } } }\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "gemini/gemini-2.5-flash", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Reply with a Greeting object whose `text` field is 'hi'.\" }],\n \"response_format\": { \"type\": \"json_schema\", \"json_schema\": { \"name\": \"Greeting\", \"strict\": true, \"schema\": { \"type\": \"object\", \"properties\": { \"text\": { \"type\": \"string\" } }, \"required\": [\"text\"], \"additionalProperties\": false } } }\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "bedrock/global.anthropic.claude-opus-4-7", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Reply with a Greeting object whose `text` field is 'hi'.\" }],\n \"response_format\": { \"type\": \"json_schema\", \"json_schema\": { \"name\": \"Greeting\", \"strict\": true, \"schema\": { \"type\": \"object\", \"properties\": { \"text\": { \"type\": \"string\" } }, \"required\": [\"text\"], \"additionalProperties\": false } } }\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } } - ] - }, - { - "name": "8.6.4 Reasoning / thinking × providers", - "description": "Reasoning/thinking activation via the request field appropriate for the model's native provider:\n- OpenAI o-series: `reasoning_effort: \"low\"`\n- Anthropic: `thinking: {type: \"enabled\", budget_tokens: 1024}`\n- Gemini 2.5+: `thinkingConfig: {thinkingBudget: 1024}` (or via Bifrost's OpenAI-shape `reasoning_effort` translation)\n\nBifrost's converter maps each field to the backend's native reasoning parameter.", - "item": [ - { "name": "8.6.4.A native chat → openai/o3-mini (reasoning_effort)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/o3-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What is 17 * 23? Think step by step.\" }],\n \"reasoning_effort\": \"low\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "8.6.4.A native chat → anthropic/claude-opus-4-7 (thinking)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What is 17 * 23? Think step by step.\" }],\n \"max_tokens\": 2048,\n \"thinking\": { \"type\": \"enabled\", \"budget_tokens\": 1024 }\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "8.6.4.A native chat → anthropic/claude-opus-4-8 (thinking)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What is 17 * 23? Think step by step.\" }],\n \"max_tokens\": 2048,\n \"thinking\": { \"type\": \"adaptive\" }\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "8.6.4.A native chat → gemini/gemini-2.5-flash (reasoning_effort translated)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What is 17 * 23? Think step by step.\" }],\n \"reasoning_effort\": \"low\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "8.6.4.E /anthropic/v1/messages → openai/gpt-5 (thinking translated)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"openai/gpt-5\",\n \"max_tokens\": 2048,\n \"messages\": [{ \"role\": \"user\", \"content\": \"What is 17 * 23? Think step by step.\" }],\n \"thinking\": { \"type\": \"enabled\", \"budget_tokens\": 1024 }\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", "host": ["{{baseUrl}}"], "path": ["anthropic", "v1", "messages"] } } }, - { "name": "8.6.4.A native chat → vertex/moonshotai/kimi-k2-thinking-maas (reasoning_effort none → dropped)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/moonshotai/kimi-k2-thinking-maas\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What is 17 * 23? Think step by step.\" }],\n \"reasoning_effort\": \"none\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "8.6.4.A native chat → vertex/minimaxai/minimax-m2-maas (reasoning_effort none → dropped)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/minimaxai/minimax-m2-maas\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What is 17 * 23? Think step by step.\" }],\n \"reasoning_effort\": \"none\"\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } } }, - { "name": "8.6.4.B native /v1/responses → vertex/moonshotai/kimi-k2-thinking-maas (reasoning none → dropped)", "request": { "method": "POST", "header": [{ "key": "Content-Type", "value": "application/json" }], "body": { "mode": "raw", "raw": "{\n \"model\": \"vertex/moonshotai/kimi-k2-thinking-maas\",\n \"input\": \"What is 17 * 23? Think step by step.\",\n \"reasoning\": { \"effort\": \"none\" }\n}" }, "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } } } - ] - }, - { - "name": "8.6.5 xAI x_search (server-side tool)", - "description": "xAI-native x_search tool tests via the native /v1/responses endpoint.\n\nThe x_search tool is a server-side search tool exclusive to xAI (grok models). When invoked the model internally calls x_semantic_search and/or x_keyword_search, returning custom_tool_call output items. The final message output_text content block carries url_citation annotations pointing at the source tweets/posts.\n\nDocs: https://docs.x.ai/developers/tools/x-search\n\nCovered here:\n 8.6.5.A – basic x_search, no extra params → verifies custom_tool_call items + final message\n 8.6.5.B – x_search with allowed_x_handles filter → tool_choice=required, verifies sub-tool name\n 8.6.5.C – x_search with from_date/to_date → date-range filtering passes through\n 8.6.5.D – x_search with all optional params (exact bug-report repro)\n 8.6.5.E – x_search streaming → custom_tool_call_input events flow through the SSE stream\n 8.6.5.F – x_search with url_citation annotations → verifies annotations array in output_text block", - "item": [ { - "name": "8.6.5.A x_search basic (no params)", - "event": [ + "name": "OpenAI (10 latest)", + "item": [ { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "pm.test('x_search: output contains at least one custom_tool_call', function () {", - " var j = pm.response.json();", - " var calls = (j.output || []).filter(function(o) { return o.type === 'custom_tool_call'; });", - " pm.expect(calls.length, 'expected ≥1 custom_tool_call in output').to.be.above(0);", - " calls.forEach(function(c) {", - " pm.expect(c.name, 'sub-tool name must start with x_').to.match(/^x_/);", - " });", - "});", - "pm.test('x_search: output contains a final message with text', function () {", - " var j = pm.response.json();", - " var msg = (j.output || []).find(function(o) { return o.type === 'message'; });", - " pm.expect(msg, 'expected a message item in output').to.exist;", - " var hasText = (msg.content || []).some(function(c) { return c.type === 'output_text' && c.text && c.text.length > 0; });", - " pm.expect(hasText, 'message should contain non-empty output_text').to.be.true;", - "});" - ] + "name": "openai/gpt-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } } - } - ], - "request": { - "method": "POST", - "header": [{ "key": "Content-Type", "value": "application/json" }], - "body": { - "mode": "raw", - "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"What are people saying about artificial intelligence on X today?\",\n \"tools\": [{ \"type\": \"x_search\" }],\n \"max_output_tokens\": 500\n}" }, - "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } - } - }, - { - "name": "8.6.5.B x_search with allowed_x_handles (tool_choice=required)", - "event": [ { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "pm.test('x_search handles: at least one custom_tool_call present', function () {", - " var j = pm.response.json();", - " var calls = (j.output || []).filter(function(o) { return o.type === 'custom_tool_call'; });", - " pm.expect(calls.length, 'tool_choice=required must produce ≥1 custom_tool_call').to.be.above(0);", - "});" - ] + "name": "openai/gpt-5-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } } - } - ], - "request": { - "method": "POST", - "header": [{ "key": "Content-Type", "value": "application/json" }], - "body": { - "mode": "raw", - "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"What has xAI been posting about recently?\",\n \"tools\": [{ \"type\": \"x_search\", \"allowed_x_handles\": [\"xai\", \"grok\"] }],\n \"tool_choice\": \"required\",\n \"max_output_tokens\": 500\n}" }, - "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } - } - }, - { - "name": "8.6.5.C x_search with from_date/to_date", - "event": [ { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "pm.test('x_search date range: custom_tool_call present', function () {", - " var j = pm.response.json();", - " var calls = (j.output || []).filter(function(o) { return o.type === 'custom_tool_call'; });", - " pm.expect(calls.length, 'expected ≥1 custom_tool_call').to.be.above(0);", - "});", - "pm.test('x_search date range: final message has content', function () {", - " var j = pm.response.json();", - " var msg = (j.output || []).find(function(o) { return o.type === 'message'; });", - " pm.expect(msg, 'expected a message item').to.exist;", - "});" - ] + "name": "openai/gpt-5-nano", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5-nano\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } } - } - ], - "request": { - "method": "POST", - "header": [{ "key": "Content-Type", "value": "application/json" }], - "body": { - "mode": "raw", - "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"What were people saying about machine learning on X recently?\",\n \"tools\": [{ \"type\": \"x_search\", \"from_date\": \"2025-12-01\", \"to_date\": \"2025-12-15\" }],\n \"max_output_tokens\": 500\n}" }, - "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } - } - }, - { - "name": "8.6.5.D x_search all optional params (bug-report repro)", - "event": [ { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "pm.test('x_search all params: tool present in request and custom_tool_call in output', function () {", - " var j = pm.response.json();", - " pm.expect(j.output, 'output should be a non-empty array').to.be.an('array').with.length.above(0);", - " var calls = (j.output || []).filter(function(o) { return o.type === 'custom_tool_call'; });", - " pm.expect(calls.length, 'tool_choice=required must yield ≥1 custom_tool_call').to.be.above(0);", - "});", - "pm.test('x_search all params: usage reports x_search_calls > 0', function () {", - " var j = pm.response.json();", - " pm.expect(j.usage, 'response must include usage').to.exist;", - " if (j.usage.server_side_tool_usage_details) {", - " pm.expect(j.usage.server_side_tool_usage_details.x_search_calls, 'x_search_calls should be > 0').to.be.above(0);", - " }", - "});" - ] + "name": "openai/gpt-4.1", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4.1\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } } - } - ], - "request": { - "method": "POST", - "header": [{ "key": "Content-Type", "value": "application/json" }], - "body": { - "mode": "raw", - "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"Find recent tweets about artificial intelligence developments.\",\n \"tools\": [{\n \"type\": \"x_search\",\n \"allowed_x_handles\": [\"xai\", \"openai\", \"GoogleAI\"],\n \"from_date\": \"2025-12-10\",\n \"to_date\": \"2025-12-15\",\n \"enable_image_understanding\": false,\n \"enable_video_understanding\": false\n }],\n \"tool_choice\": \"required\",\n \"max_output_tokens\": 500\n}" }, - "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } - } - }, - { - "name": "8.6.5.E x_search streaming", - "event": [ { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "// Streaming responses return text/event-stream — the collection-level content check skips those.", - "// We verify only that Bifrost emits a 200 and the SSE prefix.", - "pm.test('x_search stream: 200 with SSE content-type', function () {", - " pm.expect(pm.response.code).to.equal(200);", - " var ct = pm.response.headers.get('content-type') || '';", - " pm.expect(ct, 'expected text/event-stream').to.include('text/event-stream');", - "});", - "pm.test('x_search stream: body includes x_search tool event', function () {", - " var body = pm.response.text() || '';", - " var hasToolEvent = body.indexOf('custom_tool_call_input') !== -1 || body.indexOf('\"type\":\"custom_tool_call\"') !== -1;", - " pm.expect(hasToolEvent, 'expected x_search tool event (custom_tool_call_input or custom_tool_call) in SSE body').to.be.true;", - "});" - ] + "name": "openai/gpt-4.1-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4.1-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } } - } - ], - "request": { - "method": "POST", - "header": [{ "key": "Content-Type", "value": "application/json" }], - "body": { - "mode": "raw", - "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"What are people saying about xAI on X?\",\n \"tools\": [{ \"type\": \"x_search\" }],\n \"tool_choice\": \"required\",\n \"max_output_tokens\": 500,\n \"stream\": true\n}" }, - "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } - } - }, - { - "name": "8.6.5.F x_search url_citation annotations", - "event": [ { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "// x_search grounds its answer in real X posts and surfaces them as url_citation", - "// annotations on the output_text content block. Verify the annotation array is", - "// present and each entry has the required url/start_index/end_index fields.", - "pm.test('x_search citations: output_text block has url_citation annotations', function () {", - " var j = pm.response.json();", - " var msg = (j.output || []).find(function(o) { return o.type === 'message'; });", - " pm.expect(msg, 'output must contain a message item').to.exist;", - " var textBlock = (msg.content || []).find(function(c) { return c.type === 'output_text'; });", - " pm.expect(textBlock, 'message must have an output_text block').to.exist;", - " pm.expect(textBlock.annotations, 'output_text must have annotations array').to.be.an('array').with.length.above(0);", - " textBlock.annotations.forEach(function(a) {", - " pm.expect(a.type, 'annotation type must be url_citation').to.equal('url_citation');", - " pm.expect(a.url, 'url_citation must have a url').to.be.a('string').with.length.above(0);", - " pm.expect(a).to.have.property('start_index');", - " pm.expect(a).to.have.property('end_index');", - " });", - "});" - ] + "name": "openai/gpt-4.1-nano", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4.1-nano\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } } - } - ], - "request": { - "method": "POST", - "header": [{ "key": "Content-Type", "value": "application/json" }], - "body": { - "mode": "raw", - "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"What are people saying about xAI on X? Summarise with sources.\",\n \"tools\": [{ \"type\": \"x_search\" }],\n \"tool_choice\": \"required\",\n \"max_output_tokens\": 600\n}" }, - "url": { "raw": "{{baseUrl}}/v1/responses", "host": ["{{baseUrl}}"], "path": ["v1", "responses"] } - } - } - ] - } - ] - } - ] - }, - { - "name": "9. Passthrough (catch-all forwarding)", - "description": "Catch-all routes that forward the request body byte-for-byte to the upstream provider, with Bifrost injecting its own configured key. The handler at transports/bifrost-http/integrations/router.go:604 strips Authorization / API-Key / X-API-Key / X-Goog-API-Key from the inbound request - so these requests intentionally omit those headers. Bedrock and Vertex have no passthrough variant (AWS SigV4 / Google OAuth can't be bridged this way).", - "item": [ + { + "name": "openai/gpt-4o", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "openai/o3", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/o3\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "openai/o3-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/o3-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Anthropic (10 latest)", + "item": [ + { + "name": "anthropic/claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-sonnet-4-6", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-opus-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-sonnet-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-opus-4-6", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-6\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-sonnet-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-opus-4-1", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-1\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-haiku-4-5-20251001", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5-20251001\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Bedrock (10 latest)", + "item": [ + { + "name": "bedrock/global.anthropic.claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-sonnet-4-6", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-opus-4-5-20251101-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-5-20251101-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-sonnet-4-5-20250929-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-5-20250929-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/us.amazon.nova-pro-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-pro-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/us.amazon.nova-lite-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/us.amazon.nova-micro-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-micro-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/us.meta.llama3-1-70b-instruct-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.meta.llama3-1-70b-instruct-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Bedrock Mantle (10 latest)", + "item": [ + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Gemini / GenAI (10 latest)", + "item": [ + { + "name": "gemini/gemini-2.5-pro", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash-lite", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash-lite\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-3.1-pro-preview", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-3.1-pro-preview\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash-lite", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash-lite\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-3-flash-preview", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-3-flash-preview\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-3.1-flash-lite-preview", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-3.1-flash-lite-preview\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-flash-latest", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-flash-latest\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-pro-latest", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-pro-latest\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Vertex (10 latest)", + "item": [ + { + "name": "vertex/gemini-2.5-pro", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "vertex/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "vertex/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "vertex/claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "vertex/gemini-2.5-flash-lite", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash-lite\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "[PREVIEW] vertex/gemini-3-flash-preview", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-3-flash-preview\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "[PREVIEW] vertex/gemini-3.1-pro-preview", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-3.1-pro-preview\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "[PREVIEW] vertex/gemini-pro-latest", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-pro-latest\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "[PREVIEW] vertex/gemini-flash-latest", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-flash-latest\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "[PREVIEW] vertex/gemini-3.1-flash-lite-preview", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-3.1-flash-lite-preview\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + } + ] + }, + { + "name": "8.1.B Native /v1/responses (Responses shape × all providers)", + "description": "OpenAI Responses-API request shape (`input` instead of `messages`) against the native endpoint with model strings from all 6 providers. Bifrost's Responses-shape converter normalizes to BifrostChatRequest internally; the same converter must handle every backend.", + "item": [ + { + "name": "openai/gpt-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "anthropic/claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-pro", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "vertex/gemini-2.5-pro", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "vertex/claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock/us.amazon.nova-lite-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "azure/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "azure/gpt-4o", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-4o\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "xai/grok-4-0709", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + } + ] + }, + { + "name": "8.1.C OpenAI drop-in /openai/v1/chat/completions × non-OpenAI providers", + "description": "OpenAI ChatCompletion shape via the `/openai` drop-in prefix, routed to non-OpenAI backends via `provider/` model prefix. The OpenAI diagonal is already in §2 and 8.1.A; here we exercise the drop-in's converter against backends that need shape translation.", + "item": [ + { + "name": "anthropic/claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-pro", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "vertex/gemini-2.5-pro", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "vertex/claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/us.amazon.nova-lite-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "azure/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "azure/gpt-4o", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-4o\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "8.1.D OpenAI drop-in /openai/v1/responses × non-OpenAI providers", + "description": "OpenAI Responses-API shape via `/openai` drop-in, routed to non-OpenAI backends. Responses-shape converter must round-trip through every provider's chat backend.", + "item": [ + { + "name": "anthropic/claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-pro", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "vertex/gemini-2.5-pro", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "vertex/claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock/us.amazon.nova-lite-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "azure/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "azure/gpt-4o", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-4o\",\n \"input\": \"Hello\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + } + ] + }, + { + "name": "8.1.E Anthropic drop-in /anthropic/v1/messages × non-Anthropic providers", + "description": "Anthropic Messages-API shape (requires `max_tokens`) routed to non-Anthropic backends. Tests Bifrost's Messages→BifrostChatRequest converter against every other provider.", + "item": [ + { + "name": "openai/gpt-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-pro", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "vertex/gemini-2.5-pro", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "vertex/claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "bedrock/us.amazon.nova-lite-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "azure/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "azure/gpt-4o", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-4o\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Hello\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + } + ] + }, + { + "name": "8.1.F GenAI drop-in /genai/v1beta/models/{model}:generateContent × non-Gemini providers", + "description": "Google GenAI generateContent shape (`contents` / `parts` body, model in URL path) routed to non-Gemini backends. The provider/model prefix is part of the path; Bifrost's GenAI router (`integrations/router.go`) parses it before invoking the backend.", + "item": [ + { + "name": "openai/gpt-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/openai/gpt-5:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "openai", + "gpt-5:generateContent" + ] + } + } + }, + { + "name": "openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/openai/gpt-4o-mini:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "openai", + "gpt-4o-mini:generateContent" + ] + } + } + }, + { + "name": "anthropic/claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/anthropic/claude-opus-4-7:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "anthropic", + "claude-opus-4-7:generateContent" + ] + } + } + }, + { + "name": "anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/anthropic/claude-haiku-4-5:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "anthropic", + "claude-haiku-4-5:generateContent" + ] + } + } + }, + { + "name": "vertex/gemini-2.5-pro", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/vertex/gemini-2.5-pro:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "vertex", + "gemini-2.5-pro:generateContent" + ] + } + } + }, + { + "name": "vertex/claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/vertex/claude-opus-4-7:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "vertex", + "claude-opus-4-7:generateContent" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/bedrock/global.anthropic.claude-opus-4-7:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "bedrock", + "global.anthropic.claude-opus-4-7:generateContent" + ] + } + } + }, + { + "name": "bedrock/us.amazon.nova-lite-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/bedrock/us.amazon.nova-lite-v1:0:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "bedrock", + "us.amazon.nova-lite-v1:0:generateContent" + ] + } + } + }, + { + "name": "azure/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/azure/gpt-4o-mini:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "azure", + "gpt-4o-mini:generateContent" + ] + } + } + }, + { + "name": "azure/gpt-4o", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Hello\" }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/azure/gpt-4o:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "azure", + "gpt-4o:generateContent" + ] + } + } + } + ] + } + ] + }, + { + "name": "8.2 Text Chat (streaming)", + "description": "Streaming variants of the 8.1 matrix. OpenAI / Anthropic / Responses shapes use `stream: true` in the body; GenAI uses `:streamGenerateContent?alt=sse`; Bedrock uses the `/converse-stream` URL.\n\nThe folder-level test script (inherited by all items) asserts the Content-Type is a streaming type and that the body contains a recognized stream terminator (`[DONE]` for OpenAI SSE, `message_stop` / `message_delta` for Anthropic, `finishReason` for GenAI, `messageStop` for Bedrock).", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Streaming-specific assertion: header + full body consume + terminator check.", + "// Runs in addition to the collection-level 2xx check. The collection-level content-shape test skips on streaming responses (see line ~34), so this fills the gap.", + "if (pm.response.code >= 400) { return; }", + "pm.test('Streaming response with terminator', function () {", + " var ct = (pm.response.headers.get('content-type') || '').toLowerCase();", + " var isStreamCt = ct.indexOf('event-stream') !== -1 || ct.indexOf('eventstream') !== -1 || ct.indexOf('x-ndjson') !== -1;", + " pm.expect(isStreamCt, 'expected streaming content-type, got: ' + ct).to.be.true;", + " var body = pm.response.text() || '';", + " pm.expect(body.length, 'streaming body was empty').to.be.greaterThan(0);", + " var terms = ['[DONE]', 'message_stop', '\"type\":\"message_delta\"', 'response.completed', 'finishReason', 'messageStop', '\"event\":\"messageStop\"'];", + " var hit = terms.some(function (p) { return body.indexOf(p) !== -1; });", + " pm.expect(hit, 'no recognized stream terminator. First 200 chars: ' + body.slice(0, 200)).to.be.true;", + "});" + ] + } + } + ], + "item": [ + { + "name": "8.2.A Native /v1/chat/completions streaming × all providers", + "description": "Streaming chat completions via the unified native endpoint, one model per provider.", + "item": [ + { + "name": "openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "vertex/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/us.amazon.nova-lite-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "azure/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "8.2.B Native /v1/responses streaming × all providers", + "description": "Streaming Responses-API via native endpoint.", + "item": [ + { + "name": "openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "vertex/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock/us.amazon.nova-lite-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "azure/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "xai/grok-4-0709", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + } + ] + }, + { + "name": "8.2.C OpenAI drop-in /openai/v1/chat/completions streaming × non-OpenAI providers", + "description": "OpenAI streaming chat shape via drop-in, routed to non-OpenAI backends.", + "item": [ + { + "name": "anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "vertex/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/us.amazon.nova-lite-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "azure/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "8.2.D OpenAI drop-in /openai/v1/responses streaming × non-OpenAI providers", + "description": "OpenAI streaming Responses shape via drop-in, routed to non-OpenAI backends.", + "item": [ + { + "name": "anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "vertex/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock/us.amazon.nova-lite-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "azure/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"input\": \"Count to 5\",\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + } + ] + }, + { + "name": "8.2.E Anthropic drop-in /anthropic/v1/messages streaming × non-Anthropic providers", + "description": "Anthropic streaming Messages-API shape via drop-in, routed to non-Anthropic backends.", + "item": [ + { + "name": "openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "vertex/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "bedrock/us.amazon.nova-lite-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "azure/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-4o-mini\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": \"Count to 5\" }],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + } + ] + }, + { + "name": "8.2.F GenAI drop-in :streamGenerateContent × non-Gemini providers", + "description": "Google GenAI streaming via `:streamGenerateContent?alt=sse`, routed to non-Gemini backends. Body unchanged from non-streaming generateContent; SSE is requested via the URL suffix + query.", + "item": [ + { + "name": "openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Count to 5\" }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/openai/gpt-4o-mini:streamGenerateContent?alt=sse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "openai", + "gpt-4o-mini:streamGenerateContent" + ], + "query": [ + { + "key": "alt", + "value": "sse" + } + ] + } + } + }, + { + "name": "anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Count to 5\" }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/anthropic/claude-haiku-4-5:streamGenerateContent?alt=sse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "anthropic", + "claude-haiku-4-5:streamGenerateContent" + ], + "query": [ + { + "key": "alt", + "value": "sse" + } + ] + } + } + }, + { + "name": "vertex/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Count to 5\" }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/vertex/gemini-2.5-flash:streamGenerateContent?alt=sse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "vertex", + "gemini-2.5-flash:streamGenerateContent" + ], + "query": [ + { + "key": "alt", + "value": "sse" + } + ] + } + } + }, + { + "name": "bedrock/us.amazon.nova-lite-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Count to 5\" }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/bedrock/us.amazon.nova-lite-v1:0:streamGenerateContent?alt=sse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "bedrock", + "us.amazon.nova-lite-v1:0:streamGenerateContent" + ], + "query": [ + { + "key": "alt", + "value": "sse" + } + ] + } + } + }, + { + "name": "azure/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Count to 5\" }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/azure/gpt-4o-mini:streamGenerateContent?alt=sse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "azure", + "gpt-4o-mini:streamGenerateContent" + ], + "query": [ + { + "key": "alt", + "value": "sse" + } + ] + } + } + } + ] + } + ] + }, + { + "name": "8.3 Embeddings", + "description": "Embedding endpoints × providers. Anthropic returns NewUnsupportedOperationError for Embedding (core/providers/anthropic/anthropic.go:1878-1923) — those cells are tagged `[SKIP]` and skipped unless `include_skip=1`.\n\nResponse shapes (validated by extended top-level detector):\n- OpenAI: `{data: [{embedding: [...]}]}` → list-data branch\n- GenAI: `{embedding: {values: [...]}}` → genai-embedding branch\n- Bedrock Titan: `{embedding: [...]}` → bedrock-titan-embedding branch\n- Vertex: `{predictions: [{embeddings: {values}}]}` → vertex-predictions branch", + "item": [ + { + "name": "8.3.H Native /v1/embeddings × all providers", + "description": "Native embeddings endpoint with model strings from every provider. OpenAI-shape request, but Bifrost routes to the model's actual provider.", + "item": [ + { + "name": "openai/text-embedding-3-small", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/text-embedding-3-small\",\n \"input\": \"Hello world\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/embeddings", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "embeddings" + ] + } + } + }, + { + "name": "[SKIP] anthropic/claude-haiku-4-5 (no embedding endpoint)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Hello world\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/embeddings", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "embeddings" + ] + } + } + }, + { + "name": "gemini/gemini-embedding-001", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-embedding-001\",\n \"input\": \"Hello world\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/embeddings", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "embeddings" + ] + } + } + }, + { + "name": "vertex/text-embedding-005", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/text-embedding-005\",\n \"input\": \"Hello world\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/embeddings", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "embeddings" + ] + } + } + }, + { + "name": "bedrock/amazon.titan-embed-text-v2:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/amazon.titan-embed-text-v2:0\",\n \"input\": \"Hello world\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/embeddings", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "embeddings" + ] + } + } + }, + { + "name": "azure/text-embedding-ada-002", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/text-embedding-ada-002\",\n \"input\": \"Hello world\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/embeddings", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "embeddings" + ] + } + } + } + ] + }, + { + "name": "8.3.I OpenAI drop-in /openai/v1/embeddings × non-OpenAI providers", + "description": "OpenAI Embeddings shape via drop-in, routed to non-OpenAI backends.", + "item": [ + { + "name": "[SKIP] anthropic/claude-haiku-4-5 (no embedding endpoint)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Hello world\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/embeddings", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "embeddings" + ] + } + } + }, + { + "name": "gemini/gemini-embedding-001", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-embedding-001\",\n \"input\": \"Hello world\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/embeddings", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "embeddings" + ] + } + } + }, + { + "name": "vertex/text-embedding-005", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/text-embedding-005\",\n \"input\": \"Hello world\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/embeddings", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "embeddings" + ] + } + } + }, + { + "name": "bedrock/amazon.titan-embed-text-v2:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/amazon.titan-embed-text-v2:0\",\n \"input\": \"Hello world\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/embeddings", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "embeddings" + ] + } + } + }, + { + "name": "azure/text-embedding-ada-002", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/text-embedding-ada-002\",\n \"input\": \"Hello world\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/embeddings", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "embeddings" + ] + } + } + } + ] + }, + { + "name": "8.3.J GenAI drop-in /genai/v1beta/models/{model}:embedContent × non-Gemini providers", + "description": "Google GenAI embedContent shape (`content.parts`), model in URL path. Routed to non-Gemini embedding backends.", + "item": [ + { + "name": "openai/text-embedding-3-small", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"content\": { \"parts\": [{ \"text\": \"Hello world\" }] }\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/openai/text-embedding-3-small:embedContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "openai", + "text-embedding-3-small:embedContent" + ] + } + } + }, + { + "name": "[SKIP] anthropic/claude-haiku-4-5 (no embedding endpoint)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"content\": { \"parts\": [{ \"text\": \"Hello world\" }] }\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/anthropic/claude-haiku-4-5:embedContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "anthropic", + "claude-haiku-4-5:embedContent" + ] + } + } + }, + { + "name": "vertex/text-embedding-005", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"content\": { \"parts\": [{ \"text\": \"Hello world\" }] }\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/vertex/text-embedding-005:embedContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "vertex", + "text-embedding-005:embedContent" + ] + } + } + }, + { + "name": "bedrock/amazon.titan-embed-text-v2:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"content\": { \"parts\": [{ \"text\": \"Hello world\" }] }\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/bedrock/amazon.titan-embed-text-v2:0:embedContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "bedrock", + "amazon.titan-embed-text-v2:0:embedContent" + ] + } + } + }, + { + "name": "azure/text-embedding-ada-002", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"content\": { \"parts\": [{ \"text\": \"Hello world\" }] }\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/azure/text-embedding-ada-002:embedContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "azure", + "text-embedding-ada-002:embedContent" + ] + } + } + } + ] + } + ] + }, + { + "name": "8.4 Audio (transcription + speech)", + "description": "Audio endpoints × providers. Most providers don't expose a dedicated audio endpoint through Bifrost — those cells are `[SKIP]`:\n- Anthropic: no audio (core/providers/anthropic/anthropic.go:1878-1923)\n- Vertex: no transcription (core/providers/vertex/vertex.go:1722-1729)\n- Bedrock: no transcription (core/providers/bedrock/bedrock.go:1936-1941); audio input rejected (core/providers/bedrock/utils.go:1058)\n- Gemini/Vertex TTS: no /v1/audio/speech equivalent\n\nTranscription tests require `tests/e2e/api/fixtures/sample.mp3` (small audio sample) at runtime — see Phase 7 verification.\n\nResponse shapes (validated by extended top-level detector):\n- Transcription JSON: `{text: \"...\"}` → audio-transcription branch\n- TTS binary: `audio/*` Content-Type → prelude's non-JSON early-return passes status-only check", + "item": [ + { + "name": "8.4.K Native /v1/audio/transcriptions × providers", + "description": "Transcription endpoint with multipart formdata. Supported providers: OpenAI (Whisper, gpt-4o-mini-transcribe), Gemini (via generateContent under the hood), Azure (Whisper deployment).", + "item": [ + { + "name": "openai/whisper-1", + "request": { + "method": "POST", + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "model", + "value": "openai/whisper-1", + "type": "text" + }, + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.mp3", + "type": "file" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/v1/audio/transcriptions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "audio", + "transcriptions" + ] + } + } + }, + { + "name": "[SKIP] anthropic/claude-haiku-4-5 (no transcription endpoint)", + "request": { + "method": "POST", + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "model", + "value": "anthropic/claude-haiku-4-5", + "type": "text" + }, + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.mp3", + "type": "file" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/v1/audio/transcriptions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "audio", + "transcriptions" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "model", + "value": "gemini/gemini-2.5-flash", + "type": "text" + }, + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.mp3", + "type": "file" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/v1/audio/transcriptions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "audio", + "transcriptions" + ] + } + } + }, + { + "name": "[SKIP] vertex/gemini-2.5-flash (no transcription endpoint in Bifrost)", + "request": { + "method": "POST", + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "model", + "value": "vertex/gemini-2.5-flash", + "type": "text" + }, + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.mp3", + "type": "file" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/v1/audio/transcriptions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "audio", + "transcriptions" + ] + } + } + }, + { + "name": "[SKIP] bedrock/us.amazon.nova-lite-v1:0 (no transcription endpoint in Bifrost)", + "request": { + "method": "POST", + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "model", + "value": "bedrock/us.amazon.nova-lite-v1:0", + "type": "text" + }, + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.mp3", + "type": "file" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/v1/audio/transcriptions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "audio", + "transcriptions" + ] + } + } + }, + { + "name": "azure/gpt-4o-transcribe", + "request": { + "method": "POST", + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "model", + "value": "azure/gpt-4o-transcribe", + "type": "text" + }, + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.mp3", + "type": "file" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/v1/audio/transcriptions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "audio", + "transcriptions" + ] + } + } + } + ] + }, + { + "name": "8.4.L OpenAI drop-in /openai/v1/audio/transcriptions × non-OpenAI providers", + "description": "OpenAI Transcriptions shape via drop-in, routed to non-OpenAI transcription backends.", + "item": [ + { + "name": "[SKIP] anthropic/claude-haiku-4-5 (no transcription endpoint)", + "request": { + "method": "POST", + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "model", + "value": "anthropic/claude-haiku-4-5", + "type": "text" + }, + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.mp3", + "type": "file" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/audio/transcriptions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "audio", + "transcriptions" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "model", + "value": "gemini/gemini-2.5-flash", + "type": "text" + }, + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.mp3", + "type": "file" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/audio/transcriptions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "audio", + "transcriptions" + ] + } + } + }, + { + "name": "[SKIP] vertex/gemini-2.5-flash (no transcription endpoint)", + "request": { + "method": "POST", + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "model", + "value": "vertex/gemini-2.5-flash", + "type": "text" + }, + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.mp3", + "type": "file" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/audio/transcriptions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "audio", + "transcriptions" + ] + } + } + }, + { + "name": "[SKIP] bedrock/us.amazon.nova-lite-v1:0 (no transcription endpoint)", + "request": { + "method": "POST", + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "model", + "value": "bedrock/us.amazon.nova-lite-v1:0", + "type": "text" + }, + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.mp3", + "type": "file" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/audio/transcriptions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "audio", + "transcriptions" + ] + } + } + }, + { + "name": "azure/gpt-4o-transcribe", + "request": { + "method": "POST", + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "model", + "value": "azure/gpt-4o-transcribe", + "type": "text" + }, + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.mp3", + "type": "file" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/audio/transcriptions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "audio", + "transcriptions" + ] + } + } + } + ] + }, + { + "name": "8.4.M Native /v1/audio/speech × providers", + "description": "Text-to-speech endpoint. Supported: OpenAI, Azure. Anthropic/Gemini/Vertex/Bedrock have no `/v1/audio/speech`-equivalent endpoint in Bifrost.", + "item": [ + { + "name": "openai/tts-1", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/tts-1\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/audio/speech", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "audio", + "speech" + ] + } + } + }, + { + "name": "[SKIP] anthropic/claude-haiku-4-5 (no TTS)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/audio/speech", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "audio", + "speech" + ] + } + } + }, + { + "name": "[SKIP] gemini/gemini-2.5-flash (no /v1/audio/speech)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/audio/speech", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "audio", + "speech" + ] + } + } + }, + { + "name": "[SKIP] vertex/gemini-2.5-flash (no /v1/audio/speech)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/audio/speech", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "audio", + "speech" + ] + } + } + }, + { + "name": "[SKIP] bedrock/us.amazon.nova-lite-v1:0 (no TTS via Bifrost)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/audio/speech", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "audio", + "speech" + ] + } + } + }, + { + "name": "azure/gpt-4o-mini-tts", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-4o-mini-tts\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/audio/speech", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "audio", + "speech" + ] + } + } + } + ] + }, + { + "name": "8.4.N OpenAI drop-in /openai/v1/audio/speech × non-OpenAI providers", + "description": "OpenAI TTS shape via drop-in. Only Azure has a comparable backend.", + "item": [ + { + "name": "[SKIP] anthropic/claude-haiku-4-5 (no TTS)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/audio/speech", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "audio", + "speech" + ] + } + } + }, + { + "name": "[SKIP] gemini/gemini-2.5-flash (no /v1/audio/speech)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/audio/speech", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "audio", + "speech" + ] + } + } + }, + { + "name": "[SKIP] vertex/gemini-2.5-flash (no /v1/audio/speech)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/audio/speech", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "audio", + "speech" + ] + } + } + }, + { + "name": "[SKIP] bedrock/us.amazon.nova-lite-v1:0 (no TTS)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/audio/speech", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "audio", + "speech" + ] + } + } + }, + { + "name": "azure/gpt-4o-mini-tts", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-4o-mini-tts\",\n \"input\": \"Hello, world.\",\n \"voice\": \"alloy\",\n \"response_format\": \"mp3\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/audio/speech", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "audio", + "speech" + ] + } + } + } + ] + } + ] + }, + { + "name": "8.5 Image generation", + "description": "Image generation endpoints × providers. Anthropic returns NewUnsupportedOperationError for ImageGeneration (core/providers/anthropic/anthropic.go:1878-1923) — those cells are `[SKIP]`.\n\nResponse shapes (validated by extended top-level detector):\n- OpenAI: `{data: [{url|b64_json}]}` → list-data branch\n- Vertex Imagen: `{predictions: [{bytesBase64Encoded}]}` → vertex-predictions branch\n- Bedrock Titan/Nova: `{images: [\"base64\"]}` → bedrock-images branch\n\nOmits /v1/images/edits — needs a binary PNG fixture; will be added later if needed.", + "item": [ + { + "name": "8.5.O Native /v1/images/generations × providers", + "description": "Image generation endpoint with model strings from all providers. OpenAI request shape; Bifrost translates to backend-specific format internally.", + "item": [ + { + "name": "openai/gpt-image-1", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-image-1\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/images/generations", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "images", + "generations" + ] + } + } + }, + { + "name": "[SKIP] anthropic/claude-haiku-4-5 (no image gen)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/images/generations", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "images", + "generations" + ] + } + } + }, + { + "name": "gemini/imagen-4.0-generate-001", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/imagen-4.0-generate-001\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/images/generations", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "images", + "generations" + ] + } + } + }, + { + "name": "vertex/imagen-4.0-generate-001", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/imagen-4.0-generate-001\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/images/generations", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "images", + "generations" + ] + } + } + }, + { + "name": "bedrock/amazon.nova-canvas-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/amazon.nova-canvas-v1:0\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/images/generations", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "images", + "generations" + ] + } + } + }, + { + "name": "azure/gpt-image-2", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-image-2\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/images/generations", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "images", + "generations" + ] + } + } + } + ] + }, + { + "name": "8.5.P OpenAI drop-in /openai/v1/images/generations × non-OpenAI providers", + "description": "OpenAI image-generation shape via drop-in, routed to non-OpenAI backends.", + "item": [ + { + "name": "[SKIP] anthropic/claude-haiku-4-5 (no image gen)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/images/generations", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "images", + "generations" + ] + } + } + }, + { + "name": "gemini/imagen-4.0-generate-001", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/imagen-4.0-generate-001\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/images/generations", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "images", + "generations" + ] + } + } + }, + { + "name": "vertex/imagen-4.0-generate-001", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/imagen-4.0-generate-001\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/images/generations", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "images", + "generations" + ] + } + } + }, + { + "name": "bedrock/amazon.nova-canvas-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/amazon.nova-canvas-v1:0\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/images/generations", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "images", + "generations" + ] + } + } + }, + { + "name": "azure/gpt-image-2", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/gpt-image-2\",\n \"prompt\": \"A simple red apple on a white background\",\n \"n\": 1,\n \"size\": \"1024x1024\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/images/generations", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "images", + "generations" + ] + } + } + } + ] + } + ] + }, + { + "name": "8.6 Feature combinations", + "description": "Feature-specific cross-shape tests: tool calling, vision (image input), JSON / structured output, reasoning / thinking.\n\nEach feature is tested across multiple endpoint shapes with multiple model providers, proving Bifrost's converters preserve feature semantics across shape boundaries (e.g., OpenAI `tools` array → Anthropic `tools` shape when routed to Claude; Anthropic `tools` shape → Gemini `function_declarations` when routed to Gemini).\n\nResponse-shape assertions (`tool_calls` / `tool_use` / `functionCall`) are already in the top-level detector.", + "item": [ + { + "name": "8.6.1 Tool calling × endpoint shapes × providers", + "description": "Function-calling request. Each shape uses its native tool schema (OpenAI `tools[].function`, Anthropic `tools[].input_schema`, GenAI `tools[].function_declarations`, Bedrock `toolConfig.tools[].toolSpec`). The converter must translate to the backend's native tool shape and translate the response back.", + "item": [ + { + "name": "8.6.1.A native chat → openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"type\": \"function\", \"function\": { \"name\": \"get_weather\", \"description\": \"Get current weather\", \"parameters\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } } }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "8.6.1.A native chat → anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"type\": \"function\", \"function\": { \"name\": \"get_weather\", \"description\": \"Get current weather\", \"parameters\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } } }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "8.6.1.A native chat → gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"type\": \"function\", \"function\": { \"name\": \"get_weather\", \"description\": \"Get current weather\", \"parameters\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } } }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "8.6.1.A native chat → bedrock/global.anthropic.claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"type\": \"function\", \"function\": { \"name\": \"get_weather\", \"description\": \"Get current weather\", \"parameters\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } } }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "8.6.1.A native chat → bedrock/global.anthropic.claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"type\": \"function\", \"function\": { \"name\": \"get_weather\", \"description\": \"Get current weather\", \"parameters\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } } }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "8.6.1.E /anthropic/v1/messages → openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"max_tokens\": 512,\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"name\": \"get_weather\", \"description\": \"Get current weather\", \"input_schema\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "8.6.1.E /anthropic/v1/messages → gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"max_tokens\": 512,\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"name\": \"get_weather\", \"description\": \"Get current weather\", \"input_schema\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "8.6.1.E /anthropic/v1/messages → bedrock/us.amazon.nova-lite-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"max_tokens\": 512,\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"name\": \"get_weather\", \"description\": \"Get current weather\", \"input_schema\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "8.6.1.E /anthropic/v1/messages → bedrock/us.amazon.nova-lite-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [{ \"role\": \"user\", \"content\": \"What's the weather in SF?\" }],\n \"tools\": [{ \"name\": \"get_weather\", \"description\": \"Get current weather\", \"input_schema\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "8.6.1.F /genai → openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"What's the weather in SF?\" }] }],\n \"tools\": [{ \"function_declarations\": [{ \"name\": \"get_weather\", \"description\": \"Get current weather\", \"parameters\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/openai/gpt-4o-mini:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "openai", + "gpt-4o-mini:generateContent" + ] + } + } + }, + { + "name": "8.6.1.F /genai → anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"What's the weather in SF?\" }] }],\n \"tools\": [{ \"function_declarations\": [{ \"name\": \"get_weather\", \"description\": \"Get current weather\", \"parameters\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/anthropic/claude-haiku-4-5:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "anthropic", + "claude-haiku-4-5:generateContent" + ] + } + } + }, + { + "name": "8.6.1.F /genai → bedrock/us.amazon.nova-lite-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"What's the weather in SF?\" }] }],\n \"tools\": [{ \"function_declarations\": [{ \"name\": \"get_weather\", \"description\": \"Get current weather\", \"parameters\": { \"type\": \"object\", \"properties\": { \"city\": { \"type\": \"string\" } }, \"required\": [\"city\"] } }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/bedrock/us.amazon.nova-lite-v1:0:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "bedrock", + "us.amazon.nova-lite-v1:0:generateContent" + ] + } + } + } + ] + }, + { + "name": "8.6.2 Vision (image input) × endpoint shapes × providers", + "description": "Vision request with a 1×1 transparent PNG (base64 inline). Each shape uses its native image-content schema:\n- OpenAI: `content: [{type:\"image_url\", image_url:{url:\"data:image/png;base64,...\"}}]`\n- Anthropic: `content: [{type:\"image\", source:{type:\"base64\", media_type, data}}]`\n- GenAI: `parts: [{inline_data:{mime_type, data}}]`\n- Bedrock: `content: [{image:{format, source:{bytes}}}]`", + "item": [ + { + "name": "8.6.2.A native chat → openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": [{ \"type\": \"text\", \"text\": \"Describe this image in one word.\" }, { \"type\": \"image_url\", \"image_url\": { \"url\": \"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "8.6.2.A native chat → anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": [{ \"type\": \"text\", \"text\": \"Describe this image in one word.\" }, { \"type\": \"image_url\", \"image_url\": { \"url\": \"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "8.6.2.A native chat → gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": [{ \"type\": \"text\", \"text\": \"Describe this image in one word.\" }, { \"type\": \"image_url\", \"image_url\": { \"url\": \"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "8.6.2.A native chat → bedrock/global.anthropic.claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": [{ \"type\": \"text\", \"text\": \"Describe this image in one word.\" }, { \"type\": \"image_url\", \"image_url\": { \"url\": \"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "8.6.2.E /anthropic/v1/messages → openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": [{ \"type\": \"text\", \"text\": \"Describe in one word.\" }, { \"type\": \"image\", \"source\": { \"type\": \"base64\", \"media_type\": \"image/png\", \"data\": \"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "8.6.2.E /anthropic/v1/messages → gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"max_tokens\": 800,\n \"messages\": [{ \"role\": \"user\", \"content\": [{ \"type\": \"text\", \"text\": \"Describe in one word.\" }, { \"type\": \"image\", \"source\": { \"type\": \"base64\", \"media_type\": \"image/png\", \"data\": \"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "8.6.2.F /genai → openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Describe in one word.\" }, { \"inline_data\": { \"mime_type\": \"image/png\", \"data\": \"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/openai/gpt-4o-mini:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "openai", + "gpt-4o-mini:generateContent" + ] + } + } + }, + { + "name": "8.6.2.F /genai → anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{ \"parts\": [{ \"text\": \"Describe in one word.\" }, { \"inline_data\": { \"mime_type\": \"image/png\", \"data\": \"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==\" } }] }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/anthropic/claude-haiku-4-5:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "anthropic", + "claude-haiku-4-5:generateContent" + ] + } + } + } + ] + }, + { + "name": "8.6.3 JSON / structured output × providers", + "description": "OpenAI `response_format: {type: \"json_schema\", json_schema: {...}}` via native chat endpoint, routed to each provider's structured-output mechanism (Anthropic tool-use forcing, Gemini responseSchema, Bedrock tool-spec).", + "item": [ + { + "name": "openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Reply with a Greeting object whose `text` field is 'hi'.\" }],\n \"response_format\": { \"type\": \"json_schema\", \"json_schema\": { \"name\": \"Greeting\", \"strict\": true, \"schema\": { \"type\": \"object\", \"properties\": { \"text\": { \"type\": \"string\" } }, \"required\": [\"text\"], \"additionalProperties\": false } } }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Reply with a Greeting object whose `text` field is 'hi'.\" }],\n \"response_format\": { \"type\": \"json_schema\", \"json_schema\": { \"name\": \"Greeting\", \"strict\": true, \"schema\": { \"type\": \"object\", \"properties\": { \"text\": { \"type\": \"string\" } }, \"required\": [\"text\"], \"additionalProperties\": false } } }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Reply with a Greeting object whose `text` field is 'hi'.\" }],\n \"response_format\": { \"type\": \"json_schema\", \"json_schema\": { \"name\": \"Greeting\", \"strict\": true, \"schema\": { \"type\": \"object\", \"properties\": { \"text\": { \"type\": \"string\" } }, \"required\": [\"text\"], \"additionalProperties\": false } } }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-opus-4-7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Reply with a Greeting object whose `text` field is 'hi'.\" }],\n \"response_format\": { \"type\": \"json_schema\", \"json_schema\": { \"name\": \"Greeting\", \"strict\": true, \"schema\": { \"type\": \"object\", \"properties\": { \"text\": { \"type\": \"string\" } }, \"required\": [\"text\"], \"additionalProperties\": false } } }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"Reply with a Greeting object whose `text` field is 'hi'.\" }],\n \"response_format\": { \"type\": \"json_schema\", \"json_schema\": { \"name\": \"Greeting\", \"strict\": true, \"schema\": { \"type\": \"object\", \"properties\": { \"text\": { \"type\": \"string\" } }, \"required\": [\"text\"], \"additionalProperties\": false } } }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "8.6.4 Reasoning / thinking × providers", + "description": "Reasoning/thinking activation via the request field appropriate for the model's native provider:\n- OpenAI o-series: `reasoning_effort: \"low\"`\n- Anthropic: `thinking: {type: \"enabled\", budget_tokens: 1024}`\n- Gemini 2.5+: `thinkingConfig: {thinkingBudget: 1024}` (or via Bifrost's OpenAI-shape `reasoning_effort` translation)\n\nBifrost's converter maps each field to the backend's native reasoning parameter.", + "item": [ + { + "name": "8.6.4.A native chat → openai/o3-mini (reasoning_effort)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/o3-mini\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What is 17 * 23? Think step by step.\" }],\n \"reasoning_effort\": \"low\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "8.6.4.A native chat → anthropic/claude-opus-4-7 (thinking)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What is 17 * 23? Think step by step.\" }],\n \"max_tokens\": 2048,\n \"thinking\": { \"type\": \"enabled\", \"budget_tokens\": 1024 }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "8.6.4.A native chat → anthropic/claude-opus-4-8 (thinking)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What is 17 * 23? Think step by step.\" }],\n \"max_tokens\": 2048,\n \"thinking\": { \"type\": \"adaptive\" }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "8.6.4.A native chat → gemini/gemini-2.5-flash (reasoning_effort translated)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What is 17 * 23? Think step by step.\" }],\n \"reasoning_effort\": \"low\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "8.6.4.E /anthropic/v1/messages → openai/gpt-5 (thinking translated)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5\",\n \"max_tokens\": 2048,\n \"messages\": [{ \"role\": \"user\", \"content\": \"What is 17 * 23? Think step by step.\" }],\n \"thinking\": { \"type\": \"enabled\", \"budget_tokens\": 1024 }\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "8.6.4.A native chat → vertex/moonshotai/kimi-k2-thinking-maas (reasoning_effort none → dropped)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/moonshotai/kimi-k2-thinking-maas\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What is 17 * 23? Think step by step.\" }],\n \"reasoning_effort\": \"none\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "8.6.4.A native chat → vertex/minimaxai/minimax-m2-maas (reasoning_effort none → dropped)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/minimaxai/minimax-m2-maas\",\n \"messages\": [{ \"role\": \"user\", \"content\": \"What is 17 * 23? Think step by step.\" }],\n \"reasoning_effort\": \"none\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "8.6.4.B native /v1/responses → vertex/moonshotai/kimi-k2-thinking-maas (reasoning none → dropped)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/moonshotai/kimi-k2-thinking-maas\",\n \"input\": \"What is 17 * 23? Think step by step.\",\n \"reasoning\": { \"effort\": \"none\" }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + } + ] + }, + { + "name": "8.6.5 xAI x_search (server-side tool)", + "description": "xAI-native x_search tool tests via the native /v1/responses endpoint.\n\nThe x_search tool is a server-side search tool exclusive to xAI (grok models). When invoked the model internally calls x_semantic_search and/or x_keyword_search, returning custom_tool_call output items. The final message output_text content block carries url_citation annotations pointing at the source tweets/posts.\n\nDocs: https://docs.x.ai/developers/tools/x-search\n\nCovered here:\n 8.6.5.A – basic x_search, no extra params → verifies custom_tool_call items + final message\n 8.6.5.B – x_search with allowed_x_handles filter → tool_choice=required, verifies sub-tool name\n 8.6.5.C – x_search with from_date/to_date → date-range filtering passes through\n 8.6.5.D – x_search with all optional params (exact bug-report repro)\n 8.6.5.E – x_search streaming → custom_tool_call_input events flow through the SSE stream\n 8.6.5.F – x_search with url_citation annotations → verifies annotations array in output_text block", + "item": [ + { + "name": "8.6.5.A x_search basic (no params)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('x_search: output contains at least one custom_tool_call', function () {", + " var j = pm.response.json();", + " var calls = (j.output || []).filter(function(o) { return o.type === 'custom_tool_call'; });", + " pm.expect(calls.length, 'expected ≥1 custom_tool_call in output').to.be.above(0);", + " calls.forEach(function(c) {", + " pm.expect(c.name, 'sub-tool name must start with x_').to.match(/^x_/);", + " });", + "});", + "pm.test('x_search: output contains a final message with text', function () {", + " var j = pm.response.json();", + " var msg = (j.output || []).find(function(o) { return o.type === 'message'; });", + " pm.expect(msg, 'expected a message item in output').to.exist;", + " var hasText = (msg.content || []).some(function(c) { return c.type === 'output_text' && c.text && c.text.length > 0; });", + " pm.expect(hasText, 'message should contain non-empty output_text').to.be.true;", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"What are people saying about artificial intelligence on X today?\",\n \"tools\": [{ \"type\": \"x_search\" }],\n \"max_output_tokens\": 500\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "8.6.5.B x_search with allowed_x_handles (tool_choice=required)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('x_search handles: at least one custom_tool_call present', function () {", + " var j = pm.response.json();", + " var calls = (j.output || []).filter(function(o) { return o.type === 'custom_tool_call'; });", + " pm.expect(calls.length, 'tool_choice=required must produce ≥1 custom_tool_call').to.be.above(0);", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"What has xAI been posting about recently?\",\n \"tools\": [{ \"type\": \"x_search\", \"allowed_x_handles\": [\"xai\", \"grok\"] }],\n \"tool_choice\": \"required\",\n \"max_output_tokens\": 500\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "8.6.5.C x_search with from_date/to_date", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('x_search date range: custom_tool_call present', function () {", + " var j = pm.response.json();", + " var calls = (j.output || []).filter(function(o) { return o.type === 'custom_tool_call'; });", + " pm.expect(calls.length, 'expected ≥1 custom_tool_call').to.be.above(0);", + "});", + "pm.test('x_search date range: final message has content', function () {", + " var j = pm.response.json();", + " var msg = (j.output || []).find(function(o) { return o.type === 'message'; });", + " pm.expect(msg, 'expected a message item').to.exist;", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"What were people saying about machine learning on X recently?\",\n \"tools\": [{ \"type\": \"x_search\", \"from_date\": \"2025-12-01\", \"to_date\": \"2025-12-15\" }],\n \"max_output_tokens\": 500\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "8.6.5.D x_search all optional params (bug-report repro)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('x_search all params: tool present in request and custom_tool_call in output', function () {", + " var j = pm.response.json();", + " pm.expect(j.output, 'output should be a non-empty array').to.be.an('array').with.length.above(0);", + " var calls = (j.output || []).filter(function(o) { return o.type === 'custom_tool_call'; });", + " pm.expect(calls.length, 'tool_choice=required must yield ≥1 custom_tool_call').to.be.above(0);", + "});", + "pm.test('x_search all params: usage reports x_search_calls > 0', function () {", + " var j = pm.response.json();", + " pm.expect(j.usage, 'response must include usage').to.exist;", + " if (j.usage.server_side_tool_usage_details) {", + " pm.expect(j.usage.server_side_tool_usage_details.x_search_calls, 'x_search_calls should be > 0').to.be.above(0);", + " }", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"Find recent tweets about artificial intelligence developments.\",\n \"tools\": [{\n \"type\": \"x_search\",\n \"allowed_x_handles\": [\"xai\", \"openai\", \"GoogleAI\"],\n \"from_date\": \"2025-12-10\",\n \"to_date\": \"2025-12-15\",\n \"enable_image_understanding\": false,\n \"enable_video_understanding\": false\n }],\n \"tool_choice\": \"required\",\n \"max_output_tokens\": 500\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "8.6.5.E x_search streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Streaming responses return text/event-stream — the collection-level content check skips those.", + "// We verify only that Bifrost emits a 200 and the SSE prefix.", + "pm.test('x_search stream: 200 with SSE content-type', function () {", + " pm.expect(pm.response.code).to.equal(200);", + " var ct = pm.response.headers.get('content-type') || '';", + " pm.expect(ct, 'expected text/event-stream').to.include('text/event-stream');", + "});", + "pm.test('x_search stream: body includes x_search tool event', function () {", + " var body = pm.response.text() || '';", + " var hasToolEvent = body.indexOf('custom_tool_call_input') !== -1 || body.indexOf('\"type\":\"custom_tool_call\"') !== -1;", + " pm.expect(hasToolEvent, 'expected x_search tool event (custom_tool_call_input or custom_tool_call) in SSE body').to.be.true;", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"What are people saying about xAI on X?\",\n \"tools\": [{ \"type\": \"x_search\" }],\n \"tool_choice\": \"required\",\n \"max_output_tokens\": 500,\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "8.6.5.F x_search url_citation annotations", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// x_search grounds its answer in real X posts and surfaces them as url_citation", + "// annotations on the output_text content block. Verify the annotation array is", + "// present and each entry has the required url/start_index/end_index fields.", + "pm.test('x_search citations: output_text block has url_citation annotations', function () {", + " var j = pm.response.json();", + " var msg = (j.output || []).find(function(o) { return o.type === 'message'; });", + " pm.expect(msg, 'output must contain a message item').to.exist;", + " var textBlock = (msg.content || []).find(function(c) { return c.type === 'output_text'; });", + " pm.expect(textBlock, 'message must have an output_text block').to.exist;", + " pm.expect(textBlock.annotations, 'output_text must have annotations array').to.be.an('array').with.length.above(0);", + " textBlock.annotations.forEach(function(a) {", + " pm.expect(a.type, 'annotation type must be url_citation').to.equal('url_citation');", + " pm.expect(a.url, 'url_citation must have a url').to.be.a('string').with.length.above(0);", + " pm.expect(a).to.have.property('start_index');", + " pm.expect(a).to.have.property('end_index');", + " });", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"xai/grok-4-0709\",\n \"input\": \"What are people saying about xAI on X? Summarise with sources.\",\n \"tools\": [{ \"type\": \"x_search\" }],\n \"tool_choice\": \"required\",\n \"max_output_tokens\": 600\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + } + ] + } + ] + } + ] + }, + { + "name": "9. Passthrough (catch-all forwarding)", + "description": "Catch-all routes that forward the request body byte-for-byte to the upstream provider, with Bifrost injecting its own configured key. The handler at transports/bifrost-http/integrations/router.go:604 strips Authorization / API-Key / X-API-Key / X-Goog-API-Key from the inbound request - so these requests intentionally omit those headers. Bedrock and Vertex have no passthrough variant (AWS SigV4 / Google OAuth can't be bridged this way).", + "item": [ + { + "name": "POST /openai_passthrough/v1/chat/completions", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [\n { \"role\": \"user\", \"content\": \"Hello via OpenAI passthrough.\" }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai_passthrough/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai_passthrough", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "POST /openai_passthrough/v1/responses", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"input\": \"Hello via OpenAI Responses passthrough.\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai_passthrough/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai_passthrough", + "v1", + "responses" + ] + } + } + }, + { + "name": "POST /anthropic_passthrough/v1/messages", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [\n { \"role\": \"user\", \"content\": \"Hello via Anthropic passthrough.\" }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic_passthrough/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic_passthrough", + "v1", + "messages" + ] + } + } + }, + { + "name": "POST /azure_passthrough/openai/deployments/{deployment}/chat/completions", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [\n { \"role\": \"user\", \"content\": \"Hello via Azure passthrough.\" }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/azure_passthrough/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "azure_passthrough", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } + ] + } + } + }, + { + "name": "POST /azure_passthrough/openai/deployments/{deployment}/chat/completions (no api-version — default injected)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [\n { \"role\": \"user\", \"content\": \"Hello via Azure passthrough without api-version.\" }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/azure_passthrough/openai/deployments/{{azureDeployment}}/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "azure_passthrough", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ] + } + } + }, + { + "name": "POST /azure_passthrough/openai/v1/responses (no api-version — preview injected)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"{{azureDeployment}}\",\n \"input\": \"Hello via Azure v1 responses passthrough without api-version.\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/azure_passthrough/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "azure_passthrough", + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "POST /azure_passthrough/openai/deployments/gpt-4o-transcribe/audio/transcriptions (no api-version — default injected)", + "request": { + "method": "POST", + "header": [], + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "model", + "value": "gpt-4o-transcribe", + "type": "text" + }, + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.mp3", + "type": "file" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/azure_passthrough/openai/deployments/gpt-4o-transcribe/audio/transcriptions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "azure_passthrough", + "openai", + "deployments", + "gpt-4o-transcribe", + "audio", + "transcriptions" + ] + } + } + }, + { + "name": "POST /genai_passthrough/v1beta/models/{model}:generateContent", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [\n { \"parts\": [{ \"text\": \"Hello via GenAI passthrough.\" }] }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai_passthrough/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai_passthrough", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + } + ] + }, + { + "name": "10. Feature Variations (per-provider)", + "description": "Provider-native feature exercises - structured output, server-side tools (web search, code execution), function calling, beta headers, vision, streaming, prompt caching, extended thinking, etc. Each sub-folder targets a single provider's shape via its drop-in route or native call.", + "item": [ + { + "name": "OpenAI Features", + "item": [ + { + "name": "Structured output (json_schema)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [\n { \"role\": \"user\", \"content\": \"Extract the city, country, and population (rough) for Tokyo.\" }\n ],\n \"response_format\": {\n \"type\": \"json_schema\",\n \"json_schema\": {\n \"name\": \"city_info\",\n \"strict\": true,\n \"schema\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": { \"type\": \"string\" },\n \"country\": { \"type\": \"string\" },\n \"population_millions\": { \"type\": \"number\" }\n },\n \"required\": [\"city\", \"country\", \"population_millions\"],\n \"additionalProperties\": false\n }\n }\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Function calling (custom tool)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [\n { \"role\": \"user\", \"content\": \"What's the weather in Paris?\" }\n ],\n \"tools\": [\n {\n \"type\": \"function\",\n \"function\": {\n \"name\": \"get_weather\",\n \"description\": \"Get current weather for a city\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": { \"city\": { \"type\": \"string\" } },\n \"required\": [\"city\"]\n }\n }\n }\n ],\n \"tool_choice\": \"auto\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Tool choice forced (required)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a random color.\"}],\n \"tools\": [{\n \"type\": \"function\",\n \"function\": {\n \"name\": \"pick_color\",\n \"parameters\": {\"type\":\"object\",\"properties\":{\"hex\":{\"type\":\"string\"}},\"required\":[\"hex\"]}\n }\n }],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Web search (Responses API)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"input\": \"What's the latest news about quantum computing this week?\",\n \"tools\": [{ \"type\": \"web_search_preview\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "Code interpreter (Responses API)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"input\": \"Plot a sine wave and tell me its period.\",\n \"tools\": [{ \"type\": \"code_interpreter\", \"container\": { \"type\": \"auto\" } }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "Vision (image_url)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\n \"role\": \"user\",\n \"content\": [\n { \"type\": \"text\", \"text\": \"Describe this image in one sentence.\" },\n { \"type\": \"image_url\", \"image_url\": { \"url\": \"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\" } }\n ]\n }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Reasoning effort (gpt-5)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"What's 17 * 23 + sqrt(144)? Show your reasoning.\"}],\n \"reasoning_effort\": \"high\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Streaming (chat completions)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count from 1 to 10.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "System message + multi-turn", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [\n { \"role\": \"system\", \"content\": \"You are a pirate. Respond in pirate speak.\" },\n { \"role\": \"user\", \"content\": \"Hello, what time is it?\" },\n { \"role\": \"assistant\", \"content\": \"Arrr, the sun be high in the sky, matey!\" },\n { \"role\": \"user\", \"content\": \"Now tell me a joke.\" }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Anthropic Features", + "item": [ + { + "name": "Tool use (custom function)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"What's the weather in Tokyo?\"}],\n \"tools\": [{\n \"name\": \"get_weather\",\n \"description\": \"Get current weather for a city\",\n \"input_schema\": {\n \"type\": \"object\",\n \"properties\": { \"city\": { \"type\": \"string\" } },\n \"required\": [\"city\"]\n }\n }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Tool choice forced (any)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a color.\"}],\n \"tools\": [{\n \"name\": \"pick_color\",\n \"input_schema\": {\"type\":\"object\",\"properties\":{\"hex\":{\"type\":\"string\"}},\"required\":[\"hex\"]}\n }],\n \"tool_choice\": { \"type\": \"any\" }\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Web search (web_search_20250305)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"What's the weather in NYC right now?\"}],\n \"tools\": [{\n \"type\": \"web_search_20250305\",\n \"name\": \"web_search\",\n \"max_uses\": 3\n }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Web search w/ dynamic filtering (web_search_20260209)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"messages\": [{\"role\":\"user\",\"content\":\"Search for AAPL and GOOGL prices, compute P/E ratio.\"}],\n \"tools\": [\n { \"type\": \"web_search_20260209\", \"name\": \"web_search\" },\n { \"type\": \"code_execution_20250522\", \"name\": \"code_execution\" }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Web search with domain filter + location", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Find recent AI papers from arxiv.\"}],\n \"tools\": [{\n \"type\": \"web_search_20250305\",\n \"name\": \"web_search\",\n \"max_uses\": 3,\n \"allowed_domains\": [\"arxiv.org\"],\n \"user_location\": { \"type\": \"approximate\", \"city\": \"San Francisco\", \"region\": \"California\", \"country\": \"US\", \"timezone\": \"America/Los_Angeles\" }\n }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Extended thinking (Sonnet 4.6, type=enabled)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 16000,\n \"thinking\": { \"type\": \"enabled\", \"budget_tokens\": 10000 },\n \"messages\": [{\"role\":\"user\",\"content\":\"Plan a 7-day trip to Japan focusing on food. Think carefully.\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Adaptive thinking (Opus 4.7)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 16000,\n \"thinking\": { \"type\": \"adaptive\" },\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve: integral of x^2 * e^(-x) dx from 0 to infinity. Show steps.\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Adaptive thinking (Opus 4.8)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 16000,\n \"thinking\": { \"type\": \"adaptive\" },\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve: integral of x^2 * e^(-x) dx from 0 to infinity. Show steps.\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Adaptive thinking (Sonnet 5)", + "request": { + "method": "POST", + "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], + "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-5\",\n \"max_tokens\": 16000,\n \"thinking\": { \"type\": \"adaptive\" },\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve: integral of x^2 * e^(-x) dx from 0 to infinity. Show steps.\"}]\n}"}, + "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + } + }, + { + "name": "Adaptive thinking + effort=high (Sonnet 5)", + "request": { + "method": "POST", + "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], + "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-5\",\n \"max_tokens\": 16000,\n \"thinking\": { \"type\": \"adaptive\" },\n \"output_config\": { \"effort\": \"high\" },\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve: integral of x^2 * e^(-x) dx from 0 to infinity. Show steps.\"}]\n}"}, + "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + } + }, + { + "name": "Prompt caching (cache_control: ephemeral)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"system\": [\n { \"type\": \"text\", \"text\": \"You are an expert legal assistant.\" },\n { \"type\": \"text\", \"text\": \"Reference doc: [imagine 1000 lines of legal text here for caching demo]\", \"cache_control\": { \"type\": \"ephemeral\" } }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Summarize the doc.\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Computer use - Sonnet 4.5 canonical (old-gen tools, computer-use-2025-01-24)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "computer-use-2025-01-24" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-sonnet-4-5\",\n \"max_tokens\": 4096,\n \"tools\": [\n { \"type\": \"computer_20250124\", \"name\": \"computer\", \"display_width_px\": 1024, \"display_height_px\": 768, \"display_number\": 1 },\n { \"type\": \"bash_20250124\", \"name\": \"bash\" },\n { \"type\": \"text_editor_20250124\", \"name\": \"str_replace_editor\" }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot of the desktop.\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Computer use - Sonnet 4.6 canonical (new-gen tools, computer-use-2025-11-24)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "computer-use-2025-11-24" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"tools\": [\n { \"type\": \"computer_20251124\", \"name\": \"computer\", \"display_width_px\": 1024, \"display_height_px\": 768, \"display_number\": 1 },\n { \"type\": \"bash_20250124\", \"name\": \"bash\" },\n { \"type\": \"text_editor_20250728\", \"name\": \"str_replace_based_edit_tool\" }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot of the desktop.\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Computer use - Sonnet 4.5 + new-gen tools (Bifrost auto-downgrades)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "computer-use-2025-01-24" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-sonnet-4-5\",\n \"max_tokens\": 4096,\n \"tools\": [\n { \"type\": \"computer_20251124\", \"name\": \"computer\", \"display_width_px\": 1024, \"display_height_px\": 768, \"display_number\": 1 },\n { \"type\": \"bash_20250124\", \"name\": \"bash\" },\n { \"type\": \"text_editor_20250728\", \"name\": \"str_replace_based_edit_tool\" }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot of the desktop.\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Computer use - Sonnet 4.6 + old-gen tools (Bifrost auto-upgrades)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "computer-use-2025-11-24" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"tools\": [\n { \"type\": \"computer_20250124\", \"name\": \"computer\", \"display_width_px\": 1024, \"display_height_px\": 768, \"display_number\": 1 },\n { \"type\": \"bash_20250124\", \"name\": \"bash\" },\n { \"type\": \"text_editor_20250124\", \"name\": \"str_replace_editor\" }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot of the desktop.\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Computer use - Sonnet 5 canonical (new-gen tools, computer-use-2025-11-24)", + "request": { + "method": "POST", + "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"computer-use-2025-11-24"}], + "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-5\",\n \"max_tokens\": 4096,\n \"tools\": [\n { \"type\": \"computer_20251124\", \"name\": \"computer\", \"display_width_px\": 1024, \"display_height_px\": 768, \"display_number\": 1 },\n { \"type\": \"bash_20250124\", \"name\": \"bash\" },\n { \"type\": \"text_editor_20250728\", \"name\": \"str_replace_based_edit_tool\" }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot of the desktop.\"}]\n}"}, + "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + } + }, + { + "name": "Computer use - Sonnet 5 + old-gen tools (Bifrost auto-upgrades)", + "request": { + "method": "POST", + "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"computer-use-2025-11-24"}], + "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-5\",\n \"max_tokens\": 4096,\n \"tools\": [\n { \"type\": \"computer_20250124\", \"name\": \"computer\", \"display_width_px\": 1024, \"display_height_px\": 768, \"display_number\": 1 },\n { \"type\": \"bash_20250124\", \"name\": \"bash\" },\n { \"type\": \"text_editor_20250124\", \"name\": \"str_replace_editor\" }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot of the desktop.\"}]\n}"}, + "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + } + }, + { + "name": "Vision (base64 image)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\n \"role\": \"user\",\n \"content\": [\n { \"type\": \"image\", \"source\": { \"type\": \"url\", \"url\": \"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\" } },\n { \"type\": \"text\", \"text\": \"Describe this image briefly.\" }\n ]\n }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Streaming (SSE)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"stream\": true,\n \"messages\": [{\"role\":\"user\",\"content\":\"Count from 1 to 10.\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Code execution — python (code_execution_20250825)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 2048,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Use Python code execution to compute the mean and population standard deviation of [2, 4, 4, 4, 5, 5, 7, 9]. Show the code.\"\n }\n ],\n \"tools\": [\n {\n \"type\": \"code_execution_20250825\",\n \"name\": \"code_execution\"\n }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code >= 400) { return; }", + "pm.test('Anthropic: code execution ran', function () {", + " var j = pm.response.json();", + " var blocks = j.content || [];", + " var names = ['code_execution','bash_code_execution','text_editor_code_execution'];", + " var stu = blocks.find(function (b) { return b.type === 'server_tool_use' && names.indexOf(b.name) >= 0; });", + " pm.expect(stu, 'no code-execution server_tool_use block').to.be.ok;", + " var rt = ['code_execution_tool_result','bash_code_execution_tool_result','text_editor_code_execution_tool_result'];", + " var res = blocks.find(function (b) { return rt.indexOf(b.type) >= 0; });", + " pm.expect(res, 'no code-execution result block').to.be.ok;", + "});" + ] + } + } + ] + }, + { + "name": "Code execution — bash sub-tool (bash_code_execution)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 2048,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Using a bash shell command (not Python), print the Python version with `python3 --version` and list the directory with `ls -la`.\"\n }\n ],\n \"tools\": [\n {\n \"type\": \"code_execution_20250825\",\n \"name\": \"code_execution\"\n }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code >= 400) { return; }", + "pm.test('Anthropic: code execution ran', function () {", + " var j = pm.response.json();", + " var blocks = j.content || [];", + " var names = ['code_execution','bash_code_execution','text_editor_code_execution'];", + " var stu = blocks.find(function (b) { return b.type === 'server_tool_use' && names.indexOf(b.name) >= 0; });", + " pm.expect(stu, 'no code-execution server_tool_use block').to.be.ok;", + " var rt = ['code_execution_tool_result','bash_code_execution_tool_result','text_editor_code_execution_tool_result'];", + " var res = blocks.find(function (b) { return rt.indexOf(b.type) >= 0; });", + " pm.expect(res, 'no code-execution result block').to.be.ok;", + "});" + ] + } + } + ] + }, + { + "name": "Code execution — text_editor (create / view / str_replace)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 2048,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Use the file editor tool to: (1) create notes.txt containing 'debug=true', (2) view it, then (3) str_replace 'debug=true' with 'debug=false'. Use the editor, not Python.\"\n }\n ],\n \"tools\": [\n {\n \"type\": \"code_execution_20250825\",\n \"name\": \"code_execution\"\n }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code >= 400) { return; }", + "pm.test('Anthropic: code execution ran', function () {", + " var j = pm.response.json();", + " var blocks = j.content || [];", + " var names = ['code_execution','bash_code_execution','text_editor_code_execution'];", + " var stu = blocks.find(function (b) { return b.type === 'server_tool_use' && names.indexOf(b.name) >= 0; });", + " pm.expect(stu, 'no code-execution server_tool_use block').to.be.ok;", + " var rt = ['code_execution_tool_result','bash_code_execution_tool_result','text_editor_code_execution_tool_result'];", + " var res = blocks.find(function (b) { return rt.indexOf(b.type) >= 0; });", + " pm.expect(res, 'no code-execution result block').to.be.ok;", + "});", + "if (pm.response.code < 400) {", + " pm.test('Anthropic: text_editor input preserved (command + path)', function () {", + " var blocks = (pm.response.json().content) || [];", + " var te = blocks.find(function (b) { return b.type === 'server_tool_use' && b.name === 'text_editor_code_execution'; });", + " if (te) { pm.expect(te.input).to.have.property('command'); pm.expect(te.input).to.have.property('path'); }", + " });", + "}" + ] + } + } + ] + }, + { + "name": "Code execution — streaming (SSE, well-formed blocks)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 2048,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Use Python to list the first 15 Fibonacci numbers and print them.\"\n }\n ],\n \"tools\": [\n {\n \"type\": \"code_execution_20250825\",\n \"name\": \"code_execution\"\n }\n ],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code >= 400) { return; }", + "pm.test('Anthropic: streaming code execution is well-formed', function () {", + " var raw = pm.response.text() || '';", + " pm.expect(raw, 'stream did not end with message_stop').to.include('message_stop');", + " pm.expect(raw, 'no code-execution server_tool_use in stream').to.include('\"type\":\"server_tool_use\"');", + " pm.expect(raw).to.match(/\"name\":\"(code_execution|bash_code_execution|text_editor_code_execution)\"/);", + " var starts = (raw.match(/event: content_block_start/g) || []).length;", + " var stops = (raw.match(/event: content_block_stop/g) || []).length;", + " pm.expect(stops, 'content_block_start/stop unbalanced ' + starts + '/' + stops).to.equal(starts);", + " var cbIdx = [], mCb, reCb = /\"content_block_start\"[^}]*?\"index\"\\s*:\\s*(\\d+)/g;", + " while ((mCb = reCb.exec(raw)) !== null) { cbIdx.push(Number(mCb[1])); }", + " var cbSorted = cbIdx.slice().sort(function (a, b) { return a - b; });", + " for (var ci = 0; ci < cbSorted.length; ci++) { pm.expect(cbSorted[ci], 'content_block_start indices not contiguous from 0: ' + JSON.stringify(cbIdx)).to.equal(ci); }", + "});" + ] + } + } + ] + }, + { + "name": "Code execution — version code_execution_20260120 (REPL persistence + PTC)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 2048,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Use Python to compute 17 * 23 and print the result.\"\n }\n ],\n \"tools\": [\n {\n \"type\": \"code_execution_20260120\",\n \"name\": \"code_execution\"\n }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code >= 400) { return; }", + "pm.test('Anthropic: code execution ran', function () {", + " var j = pm.response.json();", + " var blocks = j.content || [];", + " var names = ['code_execution','bash_code_execution','text_editor_code_execution'];", + " var stu = blocks.find(function (b) { return b.type === 'server_tool_use' && names.indexOf(b.name) >= 0; });", + " pm.expect(stu, 'no code-execution server_tool_use block').to.be.ok;", + " var rt = ['code_execution_tool_result','bash_code_execution_tool_result','text_editor_code_execution_tool_result'];", + " var res = blocks.find(function (b) { return rt.indexOf(b.type) >= 0; });", + " pm.expect(res, 'no code-execution result block').to.be.ok;", + "});" + ] + } + } + ] + }, + { + "name": "Code execution — version code_execution_20260521 (disclosed per-cell time limit)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 2048,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Use Python to compute the factorial of 10 and print it.\"\n }\n ],\n \"tools\": [\n {\n \"type\": \"code_execution_20260521\",\n \"name\": \"code_execution\"\n }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code >= 400) { return; }", + "pm.test('Anthropic: code execution ran', function () {", + " var j = pm.response.json();", + " var blocks = j.content || [];", + " var names = ['code_execution','bash_code_execution','text_editor_code_execution'];", + " var stu = blocks.find(function (b) { return b.type === 'server_tool_use' && names.indexOf(b.name) >= 0; });", + " pm.expect(stu, 'no code-execution server_tool_use block').to.be.ok;", + " var rt = ['code_execution_tool_result','bash_code_execution_tool_result','text_editor_code_execution_tool_result'];", + " var res = blocks.find(function (b) { return rt.indexOf(b.type) >= 0; });", + " pm.expect(res, 'no code-execution result block').to.be.ok;", + "});" + ] + } + } + ] + }, + { + "name": "Code execution — programmatic tool calling (custom tool + allowed_callers)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 2048,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Write Python that calls get_stock_price for 'AAPL' and 'GOOGL' in a loop and prints each result.\"\n }\n ],\n \"tools\": [\n {\n \"type\": \"code_execution_20260120\",\n \"name\": \"code_execution\"\n },\n {\n \"name\": \"get_stock_price\",\n \"description\": \"Get the current price for a stock ticker.\",\n \"input_schema\": {\n \"type\": \"object\",\n \"properties\": {\n \"ticker\": {\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"ticker\"\n ]\n },\n \"allowed_callers\": [\n \"code_execution_20260120\"\n ]\n }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code >= 400) { return; }", + "pm.test('Anthropic: programmatic tool calling accepted (allowed_callers + auto beta header)', function () {", + " var blocks = (pm.response.json().content) || [];", + " var ran = blocks.some(function (b) {", + " return (b.type === 'server_tool_use' && b.name === 'code_execution') || (b.type === 'tool_use' && b.name === 'get_stock_price');", + " });", + " pm.expect(ran, 'neither code_execution nor the custom tool was invoked').to.be.ok;", + "});" + ] + } + } + ] + } + ] + }, + { + "name": "Bedrock Features (via /v1/chat/completions w/ bedrock prefix)", + "item": [ + { + "name": "Tool use", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"What's the weather in Sydney?\"}],\n \"tools\": [{\n \"type\": \"function\",\n \"function\": {\n \"name\": \"get_weather\",\n \"parameters\": {\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}\n }\n }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "System message", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a haiku poet. Reply only in haiku.\"},\n {\"role\":\"user\",\"content\":\"Tell me about autumn.\"}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Streaming", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"List 5 popular Python libraries.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Native Bedrock Converse w/ tool", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [\n { \"role\": \"user\", \"content\": [{\"text\": \"What's the weather in Tokyo?\"}] }\n ],\n \"toolConfig\": {\n \"tools\": [{\n \"toolSpec\": {\n \"name\": \"get_weather\",\n \"description\": \"Get current weather\",\n \"inputSchema\": { \"json\": {\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]} }\n }\n }]\n },\n \"inferenceConfig\": { \"maxTokens\": 1024 }\n}" + }, + "url": { + "raw": "{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "{{bedrockModel}}", + "converse" + ] + } + } + }, + { + "name": "Native Bedrock Converse w/ system", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hello!\"}]}],\n \"system\": [{\"text\":\"You speak only in rhymes.\"}],\n \"inferenceConfig\": {\"maxTokens\": 512}\n}" + }, + "url": { + "raw": "{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "{{bedrockModel}}", + "converse" + ] + } + } + }, + { + "name": "guardrailConfig forwarded — converse (regression)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/{{bedrockModel}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"How do I make a bomb?\"}],\n \"extra_params\": {\n \"guardrailConfig\": {\n \"guardrailIdentifier\": \"{{bedrockGuardrailIdentifier}}\",\n \"guardrailVersion\": \"{{bedrockGuardrailVersion}}\",\n \"trace\": \"enabled\"\n }\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Bedrock Mantle Features (via /v1/chat/completions w/ bedrock prefix)", + "item": [ + { + "name": "Tool use", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"What's the weather in Sydney?\"}],\n \"tools\": [{\n \"type\": \"function\",\n \"function\": {\n \"name\": \"get_weather\",\n \"parameters\": {\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}\n }\n }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "System message", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a haiku poet. Reply only in haiku.\"},\n {\"role\":\"user\",\"content\":\"Tell me about autumn.\"}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Streaming", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"List 5 popular Python libraries.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "guardrailConfig forwarded — converse (regression)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"How do I make a bomb?\"}],\n \"extra_params\": {\n \"guardrailConfig\": {\n \"guardrailIdentifier\": \"{{bedrockGuardrailIdentifier}}\",\n \"guardrailVersion\": \"{{bedrockGuardrailVersion}}\",\n \"trace\": \"enabled\"\n }\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Gemini / GenAI Features", + "item": [ + { + "name": "Structured output (responseSchema)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Extract: city, country, population_millions for Paris.\"}]}],\n \"generationConfig\": {\n \"responseMimeType\": \"application/json\",\n \"responseSchema\": {\n \"type\": \"OBJECT\",\n \"properties\": {\n \"city\": { \"type\": \"STRING\" },\n \"country\": { \"type\": \"STRING\" },\n \"population_millions\": { \"type\": \"NUMBER\" }\n },\n \"required\": [\"city\",\"country\",\"population_millions\"]\n }\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Function calling (functionDeclarations)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"What's the weather in Tokyo?\"}]}],\n \"tools\": [{\n \"functionDeclarations\": [{\n \"name\": \"get_weather\",\n \"description\": \"Get current weather for a city\",\n \"parameters\": {\n \"type\": \"OBJECT\",\n \"properties\": { \"city\": { \"type\": \"STRING\" } },\n \"required\": [\"city\"]\n }\n }]\n }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Google search grounding", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"What was announced at Google I/O 2026?\"}]}],\n \"tools\": [{ \"googleSearch\": {} }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Code execution", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Compute the 50th Fibonacci number.\"}]}],\n \"tools\": [{ \"codeExecution\": {} }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Vision (inline_data)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\n \"parts\": [\n { \"text\": \"Describe this image briefly.\" },\n { \"fileData\": { \"mimeType\": \"image/jpeg\", \"fileUri\": \"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\" } }\n ]\n }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "System instruction", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"systemInstruction\": { \"parts\": [{\"text\":\"You are a friendly chef. Always recommend a recipe.\"}] },\n \"contents\": [{\"parts\":[{\"text\":\"I have eggs and bread.\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Safety settings (BLOCK_NONE)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Tell me a story about a dragon and a knight.\"}]}],\n \"safetySettings\": [\n { \"category\": \"HARM_CATEGORY_HARASSMENT\", \"threshold\": \"BLOCK_NONE\" },\n { \"category\": \"HARM_CATEGORY_HATE_SPEECH\", \"threshold\": \"BLOCK_NONE\" },\n { \"category\": \"HARM_CATEGORY_SEXUALLY_EXPLICIT\", \"threshold\": \"BLOCK_NONE\" },\n { \"category\": \"HARM_CATEGORY_DANGEROUS_CONTENT\", \"threshold\": \"BLOCK_NONE\" }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Streaming (streamGenerateContent)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Count from 1 to 10.\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:streamGenerateContent?alt=sse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:streamGenerateContent" + ], + "query": [ + { + "key": "alt", + "value": "sse" + } + ] + } + } + }, + { + "name": "Thinking config (thinking_budget)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Solve: integral of x^2 * e^(-x) dx from 0 to infinity.\"}]}],\n \"generationConfig\": {\n \"thinkingConfig\": { \"thinkingBudget\": 8000 }\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Multi-turn with assistant history", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [\n { \"role\": \"user\", \"parts\": [{\"text\":\"Hi, who won the 2024 World Series?\"}] },\n { \"role\": \"model\", \"parts\": [{\"text\":\"The Los Angeles Dodgers won the 2024 World Series.\"}] },\n { \"role\": \"user\", \"parts\": [{\"text\":\"Who was the MVP?\"}] }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + } + ] + }, + { + "name": "Vertex Features (via /genai)", + "item": [ + { + "name": "Structured output (Vertex)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Extract city/country for London.\"}]}],\n \"generationConfig\": {\n \"responseMimeType\": \"application/json\",\n \"responseSchema\": { \"type\": \"OBJECT\", \"properties\": { \"city\": { \"type\": \"STRING\" }, \"country\": { \"type\": \"STRING\" } } }\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:generateContent" + ] + } + } + }, + { + "name": "Google search grounding (Vertex)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Latest Bedrock model launches?\"}]}],\n \"tools\": [{ \"googleSearch\": {} }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:generateContent" + ] + } + } + }, + { + "name": "Function calling (Vertex)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"What's the weather in Mumbai?\"}]}],\n \"tools\": [{\n \"functionDeclarations\": [{\n \"name\": \"get_weather\",\n \"parameters\": {\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"}},\"required\":[\"city\"]}\n }]\n }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:generateContent" + ] + } + } + }, + { + "name": "Anthropic-on-Vertex (Claude tool use)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"What's the weather in Bangalore?\"}],\n \"tools\": [{\n \"type\": \"function\",\n \"function\": {\n \"name\": \"get_weather\",\n \"parameters\": {\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}\n }\n }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + } + ] + }, + { + "name": "11. Cross-Provider Feature Tests", + "description": "Same feature exercised across multiple providers via Bifrost's unified routing. Each sub-folder is one capability; each request differs only by `model` (or path) to show that Bifrost translates the request shape per-provider.", + "item": [ + { + "name": "Structured Output cross-cut", + "item": [ + { + "name": "openai/gpt-4o-mini (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\n \"type\": \"json_schema\",\n \"json_schema\": {\n \"name\": \"city\",\n \"strict\": true,\n \"schema\": {\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}\n }\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-haiku (forced tool)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Forced tool: emit_city invoked with schema-compliant arguments', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('emit_city'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); pm.expect(a).to.have.property('country').that.is.a('string'); pm.expect(a).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"tools\": [{\n \"type\": \"function\",\n \"function\": {\n \"name\": \"emit_city\",\n \"parameters\": {\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"]}\n }\n }],\n \"tool_choice\": {\"type\":\"function\",\"function\":{\"name\":\"emit_city\"}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash (responseSchema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (pp) { return pp && pp.text; }); var c = t ? t.text : ''; pm.expect(c, 'candidates[0].content.parts[*].text empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('parts.text not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Extract city/country/pop for Paris.\"}]}],\n \"generationConfig\": {\n \"responseMimeType\": \"application/json\",\n \"responseSchema\": {\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"},\"country\":{\"type\":\"STRING\"},\"pop\":{\"type\":\"NUMBER\"}}}\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/gemini-2.5-flash:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "gemini-2.5-flash:generateContent" + ] + } + } + }, + { + "name": "vertex/gemini-2.5-pro (responseSchema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (pp) { return pp && pp.text; }); var c = t ? t.text : ''; pm.expect(c, 'candidates[0].content.parts[*].text empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('parts.text not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Extract city/country/pop for Paris.\"}]}],\n \"generationConfig\": {\n \"responseMimeType\": \"application/json\",\n \"responseSchema\": {\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"},\"country\":{\"type\":\"STRING\"},\"pop\":{\"type\":\"NUMBER\"}}}\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/gemini-2.5-pro:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "gemini-2.5-pro:generateContent" + ] + } + } + } + ] + }, + { + "name": "Web Search cross-cut", + "item": [ + { + "name": "openai/gpt-4o (web_search_preview)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o\",\n \"input\": \"Latest news on AI regulation in EU.\",\n \"tools\": [{ \"type\": \"web_search_preview\" }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "anthropic/claude-opus-4-7 (web_search_20250305)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news on AI regulation in EU.\"}],\n \"tools\": [{ \"type\": \"web_search_20250305\", \"name\": \"web_search\", \"max_uses\": 3 }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "gemini (googleSearch)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Latest news on AI regulation in EU.\"}]}],\n \"tools\": [{ \"googleSearch\": {} }]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/gemini-2.5-flash:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "gemini-2.5-flash:generateContent" + ] + } + } + } + ] + }, + { + "name": "Function Calling cross-cut", + "item": [ + { + "name": "openai/gpt-4o-mini", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-haiku-4-5", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-sonnet-4-6", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-sonnet-5", + "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], + "request": { + "method": "POST", + "header": [{"key":"Content-Type","value":"application/json"}], + "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, + "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + } + }, + { + "name": "bedrock/global.anthropic.claude-sonnet-5", + "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], + "request": { + "method": "POST", + "header": [{"key":"Content-Type","value":"application/json"}], + "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, + "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + } + }, + { + "name": "gemini/gemini-2.5-flash", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Streaming cross-cut", + "item": [ + { + "name": "openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-sonnet-5", + "request": { + "method": "POST", + "header": [{"key":"Content-Type","value":"application/json"}], + "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, + "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + } + }, + { + "name": "bedrock/global.anthropic.claude-sonnet-5", + "request": { + "method": "POST", + "header": [{"key":"Content-Type","value":"application/json"}], + "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, + "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Vision cross-cut", + "item": [ + { + "name": "openai/gpt-4o-mini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"What's in this image?\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "anthropic/claude-haiku-4-5", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"What's in this image?\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "gemini/gemini-2.5-flash", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"What's in this image?\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Context Compaction cross-cut", + "item": [ + { + "name": "Compaction via native Bifrost API (openai/gpt-4o)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Compaction: object is response.compaction', function () { var j = pm.response.json(); pm.expect(j.object).to.equal('response.compaction'); }); pm.test('Compaction: output is non-empty array', function () { var j = pm.response.json(); pm.expect(j.output).to.be.an('array').and.not.empty; }); pm.test('Compaction: last output item has type response.compaction and encrypted_content', function () { var j = pm.response.json(); var last = j.output[j.output.length - 1]; pm.expect(last.type).to.equal('compaction'); pm.expect(last.encrypted_content).to.be.a('string').and.not.empty; }); pm.test('Compaction: usage is present', function () { var j = pm.response.json(); pm.expect(j.usage).to.be.an('object'); pm.expect(j.usage.input_tokens).to.be.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o\",\n \"input\": [\n {\"role\": \"user\", \"content\": \"What is the capital of France?\"},\n {\"role\": \"assistant\", \"content\": \"The capital of France is Paris.\"},\n {\"role\": \"user\", \"content\": \"What is the population of Paris?\"},\n {\"role\": \"assistant\", \"content\": \"Paris has a population of approximately 2.1 million in the city proper, and around 12 million in the greater metropolitan area.\"},\n {\"role\": \"user\", \"content\": \"What is Paris known for?\"},\n {\"role\": \"assistant\", \"content\": \"Paris is known for the Eiffel Tower, the Louvre Museum, Notre-Dame Cathedral, world-class cuisine, fashion, and its rich history as a cultural and political center of Europe.\"}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses/compact", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses", + "compact" + ] + } + } + }, + { + "name": "Compaction (OpenAI gpt-4o)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Compaction: object is response.compaction', function () { var j = pm.response.json(); pm.expect(j.object).to.equal('response.compaction'); }); pm.test('Compaction: output is non-empty array', function () { var j = pm.response.json(); pm.expect(j.output).to.be.an('array').and.not.empty; }); pm.test('Compaction: last output item has type response.compaction and encrypted_content', function () { var j = pm.response.json(); var last = j.output[j.output.length - 1]; pm.expect(last.type).to.equal('compaction'); pm.expect(last.encrypted_content).to.be.a('string').and.not.empty; }); pm.test('Compaction: usage is present', function () { var j = pm.response.json(); pm.expect(j.usage).to.be.an('object'); pm.expect(j.usage.input_tokens).to.be.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"input\": [\n {\"role\": \"user\", \"content\": \"What is the capital of France?\"},\n {\"role\": \"assistant\", \"content\": \"The capital of France is Paris.\"},\n {\"role\": \"user\", \"content\": \"What is the population of Paris?\"},\n {\"role\": \"assistant\", \"content\": \"Paris has a population of approximately 2.1 million in the city proper, and around 12 million in the greater metropolitan area.\"},\n {\"role\": \"user\", \"content\": \"What is Paris known for?\"},\n {\"role\": \"assistant\", \"content\": \"Paris is known for the Eiffel Tower, the Louvre Museum, Notre-Dame Cathedral, world-class cuisine, fashion, and its rich history as a cultural and political center of Europe.\"}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses/compact", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses", + "compact" + ] + } + } + }, + { + "name": "Compaction (Azure gpt-4o)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Compaction: object is response.compaction', function () { var j = pm.response.json(); pm.expect(j.object).to.equal('response.compaction'); }); pm.test('Compaction: output is non-empty array', function () { var j = pm.response.json(); pm.expect(j.output).to.be.an('array').and.not.empty; }); pm.test('Compaction: last output item has type response.compaction and encrypted_content', function () { var j = pm.response.json(); var last = j.output[j.output.length - 1]; pm.expect(last.type).to.equal('compaction'); pm.expect(last.encrypted_content).to.be.a('string').and.not.empty; }); pm.test('Compaction: usage is present', function () { var j = pm.response.json(); pm.expect(j.usage).to.be.an('object'); pm.expect(j.usage.input_tokens).to.be.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/{{azureDeployment}}\",\n \"input\": [\n {\"role\": \"user\", \"content\": \"What is the capital of France?\"},\n {\"role\": \"assistant\", \"content\": \"The capital of France is Paris.\"},\n {\"role\": \"user\", \"content\": \"What is the population of Paris?\"},\n {\"role\": \"assistant\", \"content\": \"Paris has a population of approximately 2.1 million in the city proper, and around 12 million in the greater metropolitan area.\"},\n {\"role\": \"user\", \"content\": \"What is Paris known for?\"},\n {\"role\": \"assistant\", \"content\": \"Paris is known for the Eiffel Tower, the Louvre Museum, Notre-Dame Cathedral, world-class cuisine, fashion, and its rich history as a cultural and political center of Europe.\"}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses/compact", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses", + "compact" + ] + } + } + }, + { + "name": "Compaction (xAI grok-4.3)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Compaction: object is response.compaction', function () { var j = pm.response.json(); pm.expect(j.object).to.equal('response.compaction'); }); pm.test('Compaction: output is non-empty array', function () { var j = pm.response.json(); pm.expect(j.output).to.be.an('array').and.not.empty; }); pm.test('Compaction: last output item has type response.compaction and encrypted_content', function () { var j = pm.response.json(); var last = j.output[j.output.length - 1]; pm.expect(last.type).to.equal('compaction'); pm.expect(last.encrypted_content).to.be.a('string').and.not.empty; }); pm.test('Compaction: usage is present', function () { var j = pm.response.json(); pm.expect(j.usage).to.be.an('object'); pm.expect(j.usage.input_tokens).to.be.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"xai/grok-4.3\",\n \"input\": [\n {\"role\": \"user\", \"content\": \"What is the capital of France?\"},\n {\"role\": \"assistant\", \"content\": \"The capital of France is Paris.\"},\n {\"role\": \"user\", \"content\": \"What is the population of Paris?\"},\n {\"role\": \"assistant\", \"content\": \"Paris has a population of approximately 2.1 million in the city proper, and around 12 million in the greater metropolitan area.\"},\n {\"role\": \"user\", \"content\": \"What is Paris known for?\"},\n {\"role\": \"assistant\", \"content\": \"Paris is known for the Eiffel Tower, the Louvre Museum, Notre-Dame Cathedral, world-class cuisine, fashion, and its rich history as a cultural and political center of Europe.\"}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses/compact", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses", + "compact" + ] + } + } + }, + { + "name": "Compaction with instructions (OpenAI gpt-4o)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Compaction: object is response.compaction', function () { var j = pm.response.json(); pm.expect(j.object).to.equal('response.compaction'); }); pm.test('Compaction: output is non-empty array', function () { var j = pm.response.json(); pm.expect(j.output).to.be.an('array').and.not.empty; }); pm.test('Compaction: last output item has type response.compaction and encrypted_content', function () { var j = pm.response.json(); var last = j.output[j.output.length - 1]; pm.expect(last.type).to.equal('compaction'); pm.expect(last.encrypted_content).to.be.a('string').and.not.empty; }); pm.test('Compaction: usage is present', function () { var j = pm.response.json(); pm.expect(j.usage).to.be.an('object'); pm.expect(j.usage.input_tokens).to.be.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"input\": [\n {\"role\": \"user\", \"content\": \"What is the capital of France?\"},\n {\"role\": \"assistant\", \"content\": \"The capital of France is Paris.\"},\n {\"role\": \"user\", \"content\": \"What is the population of Paris?\"},\n {\"role\": \"assistant\", \"content\": \"Paris has a population of approximately 2.1 million in the city proper, and around 12 million in the greater metropolitan area.\"},\n {\"role\": \"user\", \"content\": \"What is Paris known for?\"},\n {\"role\": \"assistant\", \"content\": \"Paris is known for the Eiffel Tower, the Louvre Museum, Notre-Dame Cathedral, world-class cuisine, fashion, and its rich history as a cultural and political center of Europe.\"}\n ],\n \"instructions\": \"You are a helpful geography assistant. Be concise.\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses/compact", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses", + "compact" + ] + } + } + } + ] + }, + { + "name": "MCP Tool Handling cross-cut", + "description": "Regression #3795: a provider-side `type:\"mcp\"` server tool in a Responses request must be silently dropped for providers without an MCP connector (Bedrock, Vertex), not rejected. Function tools survive the strip. Function-bearing rows force the surviving function tool via tool_choice and verify it is invoked (output function_call / streamed response.function_call_arguments). Lone-mcp rows exercise the all-tools-dropped edge case: zero tools left after the strip must still return a text answer with NO tool call. Server/MCP tool calls are NOT expected for Bedrock/Vertex Claude — that is correct. Covers native + /openai drop-in + streaming.", + "item": [ + { + "name": "bedrock/global.anthropic.claude-opus-4-7 · lone server-mcp dropped", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795 'all tools dropped' edge case: the mcp server tool is the ONLY tool, so after", + "// the strip there are zero tools left. The request must still return a normal text", + "// completion, and NO tool call may appear (the mcp tool was removed, not made callable).", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') === -1) {", + " pm.test('text answer returned, no tool call (mcp dropped, zero tools left)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'no tool call expected (mcp dropped, no function tools), got: ' + JSON.stringify(out).slice(0, 200)).to.equal(0);", + " var txt = (typeof j.output_text === 'string') ? j.output_text : '';", + " var hasText = txt.length > 0 ||", + " out.some(function (o) { return o && o.type === 'message' && o.content && o.content.length; }) ||", + " (j.output && j.output.message && j.output.message.content && j.output.message.content.length);", + " pm.expect(hasText, 'expected a text/message answer, got: ' + JSON.stringify(j).slice(0, 200)).to.be.true;", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"What is 2+2? Answer in one word.\",\n \"tools\": [\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795 'all tools dropped' edge case: the mcp server tool is the ONLY tool, so after", + "// the strip there are zero tools left. The request must still return a normal text", + "// completion, and NO tool call may appear (the mcp tool was removed, not made callable).", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') === -1) {", + " pm.test('text answer returned, no tool call (mcp dropped, zero tools left)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'no tool call expected (mcp dropped, no function tools), got: ' + JSON.stringify(out).slice(0, 200)).to.equal(0);", + " var txt = (typeof j.output_text === 'string') ? j.output_text : '';", + " var hasText = txt.length > 0 ||", + " out.some(function (o) { return o && o.type === 'message' && o.content && o.content.length; }) ||", + " (j.output && j.output.message && j.output.message.content && j.output.message.content.length);", + " pm.expect(hasText, 'expected a text/message answer, got: ' + JSON.stringify(j).slice(0, 200)).to.be.true;", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"input\": \"What is 2+2? Answer in one word.\",\n \"tools\": [\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-opus-4-7 · server-mcp + function kept (forced call)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'get_weather';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'get_weather';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-opus-4-7 · 2 server-mcp + 2 function (#3795 shape, forced call)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'set_serialviewer_pro_query';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"Build a MongoDB find filter for the most common error codes in the orders collection over the last month.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\",\n \"description\": \"Writes a MongoDB find filter.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"filter\": {\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"filter\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"time\",\n \"server_url\": \"https://bifrost.invalid/mcp/time\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"function\",\n \"name\": \"mongodb-explain\",\n \"description\": \"Returns query-plan statistics.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"collection\": {\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"additionalProperties\": false\n }\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'set_serialviewer_pro_query';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"input\": \"Build a MongoDB find filter for the most common error codes in the orders collection over the last month.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\",\n \"description\": \"Writes a MongoDB find filter.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"filter\": {\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"filter\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"time\",\n \"server_url\": \"https://bifrost.invalid/mcp/time\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"function\",\n \"name\": \"mongodb-explain\",\n \"description\": \"Returns query-plan statistics.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"collection\": {\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"additionalProperties\": false\n }\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-sonnet-4-6 · lone server-mcp dropped", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795 'all tools dropped' edge case: the mcp server tool is the ONLY tool, so after", + "// the strip there are zero tools left. The request must still return a normal text", + "// completion, and NO tool call may appear (the mcp tool was removed, not made callable).", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') === -1) {", + " pm.test('text answer returned, no tool call (mcp dropped, zero tools left)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'no tool call expected (mcp dropped, no function tools), got: ' + JSON.stringify(out).slice(0, 200)).to.equal(0);", + " var txt = (typeof j.output_text === 'string') ? j.output_text : '';", + " var hasText = txt.length > 0 ||", + " out.some(function (o) { return o && o.type === 'message' && o.content && o.content.length; }) ||", + " (j.output && j.output.message && j.output.message.content && j.output.message.content.length);", + " pm.expect(hasText, 'expected a text/message answer, got: ' + JSON.stringify(j).slice(0, 200)).to.be.true;", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"input\": \"What is 2+2? Answer in one word.\",\n \"tools\": [\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795 'all tools dropped' edge case: the mcp server tool is the ONLY tool, so after", + "// the strip there are zero tools left. The request must still return a normal text", + "// completion, and NO tool call may appear (the mcp tool was removed, not made callable).", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') === -1) {", + " pm.test('text answer returned, no tool call (mcp dropped, zero tools left)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'no tool call expected (mcp dropped, no function tools), got: ' + JSON.stringify(out).slice(0, 200)).to.equal(0);", + " var txt = (typeof j.output_text === 'string') ? j.output_text : '';", + " var hasText = txt.length > 0 ||", + " out.some(function (o) { return o && o.type === 'message' && o.content && o.content.length; }) ||", + " (j.output && j.output.message && j.output.message.content && j.output.message.content.length);", + " pm.expect(hasText, 'expected a text/message answer, got: ' + JSON.stringify(j).slice(0, 200)).to.be.true;", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"input\": \"What is 2+2? Answer in one word.\",\n \"tools\": [\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-sonnet-4-6 · server-mcp + function kept (forced call)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'get_weather';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'get_weather';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-sonnet-4-6 · 2 server-mcp + 2 function (#3795 shape, forced call)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'set_serialviewer_pro_query';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"input\": \"Build a MongoDB find filter for the most common error codes in the orders collection over the last month.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\",\n \"description\": \"Writes a MongoDB find filter.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"filter\": {\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"filter\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"time\",\n \"server_url\": \"https://bifrost.invalid/mcp/time\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"function\",\n \"name\": \"mongodb-explain\",\n \"description\": \"Returns query-plan statistics.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"collection\": {\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"additionalProperties\": false\n }\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'set_serialviewer_pro_query';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"input\": \"Build a MongoDB find filter for the most common error codes in the orders collection over the last month.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\",\n \"description\": \"Writes a MongoDB find filter.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"filter\": {\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"filter\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"time\",\n \"server_url\": \"https://bifrost.invalid/mcp/time\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"function\",\n \"name\": \"mongodb-explain\",\n \"description\": \"Returns query-plan statistics.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"collection\": {\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"additionalProperties\": false\n }\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "vertex/claude-opus-4-7 · lone server-mcp dropped", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795 'all tools dropped' edge case: the mcp server tool is the ONLY tool, so after", + "// the strip there are zero tools left. The request must still return a normal text", + "// completion, and NO tool call may appear (the mcp tool was removed, not made callable).", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') === -1) {", + " pm.test('text answer returned, no tool call (mcp dropped, zero tools left)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'no tool call expected (mcp dropped, no function tools), got: ' + JSON.stringify(out).slice(0, 200)).to.equal(0);", + " var txt = (typeof j.output_text === 'string') ? j.output_text : '';", + " var hasText = txt.length > 0 ||", + " out.some(function (o) { return o && o.type === 'message' && o.content && o.content.length; }) ||", + " (j.output && j.output.message && j.output.message.content && j.output.message.content.length);", + " pm.expect(hasText, 'expected a text/message answer, got: ' + JSON.stringify(j).slice(0, 200)).to.be.true;", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"input\": \"What is 2+2? Answer in one word.\",\n \"tools\": [\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "vertex/claude-opus-4-7 · server-mcp + function kept (forced call)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'get_weather';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "vertex/claude-opus-4-7 · 2 server-mcp + 2 function (#3795 shape, forced call)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'set_serialviewer_pro_query';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"input\": \"Build a MongoDB find filter for the most common error codes in the orders collection over the last month.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\",\n \"description\": \"Writes a MongoDB find filter.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"filter\": {\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"filter\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"time\",\n \"server_url\": \"https://bifrost.invalid/mcp/time\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"function\",\n \"name\": \"mongodb-explain\",\n \"description\": \"Returns query-plan statistics.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"collection\": {\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"additionalProperties\": false\n }\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "vertex/claude-sonnet-4-6 · lone server-mcp dropped", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795 'all tools dropped' edge case: the mcp server tool is the ONLY tool, so after", + "// the strip there are zero tools left. The request must still return a normal text", + "// completion, and NO tool call may appear (the mcp tool was removed, not made callable).", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') === -1) {", + " pm.test('text answer returned, no tool call (mcp dropped, zero tools left)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'no tool call expected (mcp dropped, no function tools), got: ' + JSON.stringify(out).slice(0, 200)).to.equal(0);", + " var txt = (typeof j.output_text === 'string') ? j.output_text : '';", + " var hasText = txt.length > 0 ||", + " out.some(function (o) { return o && o.type === 'message' && o.content && o.content.length; }) ||", + " (j.output && j.output.message && j.output.message.content && j.output.message.content.length);", + " pm.expect(hasText, 'expected a text/message answer, got: ' + JSON.stringify(j).slice(0, 200)).to.be.true;", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"input\": \"What is 2+2? Answer in one word.\",\n \"tools\": [\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "vertex/claude-sonnet-4-6 · server-mcp + function kept (forced call)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'get_weather';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "vertex/claude-sonnet-4-6 · 2 server-mcp + 2 function (#3795 shape, forced call)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'set_serialviewer_pro_query';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"input\": \"Build a MongoDB find filter for the most common error codes in the orders collection over the last month.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\",\n \"description\": \"Writes a MongoDB find filter.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"filter\": {\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"filter\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"time\",\n \"server_url\": \"https://bifrost.invalid/mcp/time\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"function\",\n \"name\": \"mongodb-explain\",\n \"description\": \"Returns query-plan statistics.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"collection\": {\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"additionalProperties\": false\n }\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-opus-4-7 · /openai drop-in · forced function call", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'get_weather';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'get_weather';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "vertex/claude-opus-4-7 · /openai drop-in · forced function call", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'get_weather';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-opus-4-7 · streaming · forced function call", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'get_weather';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n },\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock_mantle/anthropic.claude-opus-4-8", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'get_weather';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n },\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "vertex/claude-opus-4-7 · streaming · forced function call", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", + "// so a server/MCP tool call must NOT happen here — that is expected and fine.", + "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", + "var ct = (pm.response.headers.get('content-type') || '');", + "var raw = pm.response.text() || '';", + "var EXPECT = 'get_weather';", + "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "if (ct.indexOf('event-stream') !== -1) {", + " pm.test('streaming response invoked function tool ' + EXPECT, function () {", + " pm.expect(raw, 'no function_call events in SSE stream')", + " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", + " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", + " });", + "} else {", + " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", + " var names = calls.map(function (o) { return o.name; });", + " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", + " });", + "}" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n },\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-opus-4-7 · tool_choice pins absent fn + lone mcp dropped → reconciled, no 400 (PR #4573)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Fix (PR #4573, CodeRabbit): tool_choice pins a function, but the ONLY tool sent is an", + "// unsupported mcp server tool. The mcp tool is dropped -> zero tools, so the dangling", + "// pin must be reconciled away (fall back to Bedrock 'auto') instead of emitting a", + "// toolChoice.tool that references an absent tool, which Bedrock rejects with HTTP 400.", + "// The collection-level 'Status code is 2xx' is the real regression guard (pre-fix => 400).", + "var raw = pm.response.text() || '';", + "pm.test('mcp dropped + dangling tool_choice reconciled, no 400 (PR #4573)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + " pm.expect(raw.toLowerCase(), 'pinned tool must not leak to Bedrock as an unknown tool').to.not.include('tool not found');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "pm.test('no tool call (pin reconciled away, zero tools left)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected no tool call (no tools survived), got: ' + JSON.stringify(out).slice(0, 200)).to.equal(0);", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"What is 2+2? Answer in one word.\",\n \"tools\": [\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "bedrock/global.anthropic.claude-sonnet-4-6 · tool_choice pins absent fn + lone mcp dropped → reconciled, no 400 (PR #4573)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Fix (PR #4573, CodeRabbit): tool_choice pins a function, but the ONLY tool sent is an", + "// unsupported mcp server tool. The mcp tool is dropped -> zero tools, so the dangling", + "// pin must be reconciled away (fall back to Bedrock 'auto') instead of emitting a", + "// toolChoice.tool that references an absent tool, which Bedrock rejects with HTTP 400.", + "// The collection-level 'Status code is 2xx' is the real regression guard (pre-fix => 400).", + "var raw = pm.response.text() || '';", + "pm.test('mcp dropped + dangling tool_choice reconciled, no 400 (PR #4573)', function () {", + " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", + " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", + " pm.expect(raw.toLowerCase(), 'pinned tool must not leak to Bedrock as an unknown tool').to.not.include('tool not found');", + "});", + "pm.test('response body is non-empty', function () {", + " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", + "});", + "pm.test('no tool call (pin reconciled away, zero tools left)', function () {", + " var j = pm.response.json();", + " var out = Array.isArray(j.output) ? j.output : [];", + " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", + " pm.expect(calls.length, 'expected no tool call (no tools survived), got: ' + JSON.stringify(out).slice(0, 200)).to.equal(0);", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"input\": \"What is 2+2? Answer in one word.\",\n \"tools\": [\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + } + ] + } + ] + }, + { + "name": "11b. Vertex GCS Files (/openai + native resumable)", + "description": "Vertex stores files in a customer GCS bucket. CRUD runs via the OpenAI drop-in (storage_config.gcs); resumable upload runs via the native API (mint session -> client PUTs bytes straight to GCS -> cleanup). All [PREVIEW]-tagged: needs a gateway Vertex provider key (server-side) + a GCS bucket. Set vertexGcsBucket and run with --env-var include_preview=1. The OpenAI-drop-in rows send no Authorization header (routing is by provider=vertex).", + "item": [ + { + "name": "[PREVIEW] Vertex: upload file (GCS, /openai)", + "request": { + "method": "POST", + "header": [], + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.jsonl", + "type": "file" + }, + { + "key": "purpose", + "value": "batch", + "type": "text" + }, + { + "key": "provider", + "value": "vertex", + "type": "text" + }, + { + "key": "storage_config[gcs][bucket]", + "value": "{{vertexGcsBucket}}", + "type": "text" + }, + { + "key": "storage_config[gcs][prefix]", + "value": "{{vertexGcsPrefix}}", + "type": "text" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/files", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "files" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var j = pm.response.json();", + "pm.test('Vertex upload: gcs backend', function () { pm.expect(j.storage_backend).to.eql('gcs'); });", + "pm.test('Vertex upload: has id', function () { pm.expect(j.id).to.be.a('string').and.not.empty; });", + "pm.collectionVariables.set('vertexFileId', j.id);" + ] + } + } + ] + }, + { + "name": "[PREVIEW] Vertex: list files (GCS, /openai)", + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{baseUrl}}/openai/v1/files?provider=vertex&storage_config[gcs][bucket]={{vertexGcsBucket}}&storage_config[gcs][prefix]={{vertexGcsPrefix}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "files" + ], + "query": [ + { + "key": "provider", + "value": "vertex" + }, + { + "key": "storage_config[gcs][bucket]", + "value": "{{vertexGcsBucket}}" + }, + { + "key": "storage_config[gcs][prefix]", + "value": "{{vertexGcsPrefix}}" + } + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var j = pm.response.json();", + "pm.test('Vertex list: uploaded file present', function () { pm.expect((j.data || []).map(function (f) { return f.id; })).to.include(pm.collectionVariables.get('vertexFileId')); });" + ] + } + } + ] + }, + { + "name": "[PREVIEW] Vertex: retrieve file (GCS, /openai)", + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{baseUrl}}/openai/v1/files/{{vertexFileId}}?provider=vertex", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "files", + "{{vertexFileId}}" + ], + "query": [ + { + "key": "provider", + "value": "vertex" + } + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var j = pm.response.json();", + "pm.test('Vertex retrieve: id matches uploaded', function () { pm.expect(j.id).to.eql(pm.collectionVariables.get('vertexFileId')); });" + ] + } + } + ] + }, + { + "name": "[PREVIEW] Vertex: download file content (GCS, /openai)", + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{baseUrl}}/openai/v1/files/{{vertexFileId}}/content?provider=vertex", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "files", + "{{vertexFileId}}", + "content" + ], + "query": [ + { + "key": "provider", + "value": "vertex" + } + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Vertex content: non-empty body', function () { pm.expect((pm.response.text() || '').length).to.be.above(0); });" + ] + } + } + ] + }, + { + "name": "[PREVIEW] Vertex: delete file (GCS, /openai)", + "request": { + "method": "DELETE", + "header": [], + "url": { + "raw": "{{baseUrl}}/openai/v1/files/{{vertexFileId}}?provider=vertex", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "files", + "{{vertexFileId}}" + ], + "query": [ + { + "key": "provider", + "value": "vertex" + } + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var j = pm.response.json();", + "pm.test('Vertex delete: deleted true', function () { pm.expect(j.deleted).to.eql(true); });" + ] + } + } + ] + }, + { + "name": "[PREVIEW] Vertex: mint resumable session (native)", + "request": { + "method": "POST", + "header": [], + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "purpose", + "value": "user_data", + "type": "text" + }, + { + "key": "filename", + "value": "harness-video.bin", + "type": "text" + }, + { + "key": "content_type", + "value": "application/octet-stream", + "type": "text" + }, + { + "key": "gcs_bucket", + "value": "{{vertexGcsBucket}}", + "type": "text" + }, + { + "key": "gcs_prefix", + "value": "{{vertexGcsPrefix}}", + "type": "text" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/v1/files?provider=vertex", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "files" + ], + "query": [ + { + "key": "provider", + "value": "vertex" + } + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var j = pm.response.json();", + "pm.test('Vertex resumable: status pending_upload', function () { pm.expect(j.status).to.eql('pending_upload'); });", + "pm.test('Vertex resumable: has upload_url', function () { pm.expect(j.upload_url).to.be.a('string').and.not.empty; });", + "pm.collectionVariables.set('vertexUploadUrl', j.upload_url);", + "pm.collectionVariables.set('vertexResumableId', encodeURIComponent(j.id));" + ] + } + } + ] + }, + { + "name": "[PREVIEW] Vertex: PUT bytes to GCS session (resumable, direct to GCS)", + "request": { + "method": "PUT", + "header": [], + "body": { + "mode": "file", + "file": { + "src": "tests/e2e/api/fixtures/sample.txt" + } + }, + "url": { + "raw": "{{vertexUploadUrl}}", + "host": [ + "{{vertexUploadUrl}}" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Vertex resumable: GCS accepted bytes (2xx)', function () { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ] + }, + { + "name": "[PREVIEW] Vertex: delete resumable file (native, cleanup)", + "request": { + "method": "DELETE", + "header": [], + "url": { + "raw": "{{baseUrl}}/v1/files/{{vertexResumableId}}?provider=vertex", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "files", + "{{vertexResumableId}}" + ], + "query": [ + { + "key": "provider", + "value": "vertex" + } + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var j = pm.response.json();", + "pm.test('Vertex resumable cleanup: deleted', function () { pm.expect(j.deleted).to.eql(true); });" + ] + } + } + ] + } + ] + }, + { + "name": "11c. Vertex Batches (/openai + native passthrough)", + "description": "Vertex batch prediction is GCS-backed. CRUD (create/list/retrieve/cancel) runs via the OpenAI drop-in (/openai/v1/batches, routed by provider=vertex); the base64 batch ids round-trip through every endpoint. The last two rows exercise the native genai surface raw passthrough: a verbatim Vertex BatchPredictionJob body POSTed to .../batchPredictionJobs is forwarded as-is and the native job resource is returned. All [PREVIEW]-tagged: needs a gateway Vertex provider key + a GCS bucket, plus vertexProject/vertexLocation for the native rows. Set vertexGcsBucket, vertexProject and run with --env-var include_preview=1.", + "item": [ + { + "name": "[PREVIEW] Vertex: upload batch input (GCS, /openai)", + "request": { + "method": "POST", + "header": [], + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.jsonl", + "type": "file" + }, + { + "key": "purpose", + "value": "batch", + "type": "text" + }, + { + "key": "provider", + "value": "vertex", + "type": "text" + }, + { + "key": "storage_config[gcs][bucket]", + "value": "{{vertexGcsBucket}}", + "type": "text" + }, + { + "key": "storage_config[gcs][prefix]", + "value": "{{vertexGcsPrefix}}", + "type": "text" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/files", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "files" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var j = pm.response.json();", + "pm.test('Vertex batch input: gcs backend', function () { pm.expect(j.storage_backend).to.eql('gcs'); });", + "pm.test('Vertex batch input: has id', function () { pm.expect(j.id).to.be.a('string').and.not.empty; });", + "pm.collectionVariables.set('vertexBatchInputFileId', j.id);", + "pm.collectionVariables.set('vertexBatchInputGcsUri', j.storage_uri);" + ] + } + } + ] + }, + { + "name": "[PREVIEW] Vertex: create batch (/openai)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"input_file_id\": \"{{vertexBatchInputFileId}}\",\n \"endpoint\": \"/v1/chat/completions\",\n \"completion_window\": \"24h\",\n \"provider\": \"vertex\",\n \"model\": \"gemini-2.5-flash\",\n \"output_folder\": {\n \"url\": \"gs://{{vertexGcsBucket}}/{{vertexGcsPrefix}}batch-output\"\n }\n}", + "options": { + "raw": { + "language": "json" + } + } + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/batches", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "batches" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var j = pm.response.json();", + "pm.test('Vertex create batch: object batch', function () { pm.expect(j.object).to.eql('batch'); });", + "pm.test('Vertex create batch: has id', function () { pm.expect(j.id).to.be.a('string').and.not.empty; });", + "pm.test('Vertex create batch: input_file_id round-trips', function () { pm.expect(j.input_file_id).to.eql(pm.collectionVariables.get('vertexBatchInputFileId')); });", + "pm.collectionVariables.set('vertexBatchId', j.id);" + ] + } + } + ] + }, + { + "name": "[PREVIEW] Vertex: list batches (/openai)", + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{baseUrl}}/openai/v1/batches?provider=vertex", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "batches" + ], + "query": [ + { + "key": "provider", + "value": "vertex" + } + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var j = pm.response.json();", + "pm.test('Vertex list batches: created batch present', function () { pm.expect((j.data || []).map(function (b) { return b.id; })).to.include(pm.collectionVariables.get('vertexBatchId')); });" + ] + } + } + ] + }, + { + "name": "[PREVIEW] Vertex: retrieve batch (/openai)", + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{baseUrl}}/openai/v1/batches/{{vertexBatchId}}?provider=vertex", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "batches", + "{{vertexBatchId}}" + ], + "query": [ + { + "key": "provider", + "value": "vertex" + } + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var j = pm.response.json();", + "pm.test('Vertex retrieve batch: object batch', function () { pm.expect(j.object).to.eql('batch'); });", + "pm.test('Vertex retrieve batch: id matches create', function () { pm.expect(j.id).to.eql(pm.collectionVariables.get('vertexBatchId')); });" + ] + } + } + ] + }, + { + "name": "[PREVIEW] Vertex: cancel batch (/openai)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"provider\": \"vertex\"\n}", + "options": { + "raw": { + "language": "json" + } + } + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/batches/{{vertexBatchId}}/cancel", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "batches", + "{{vertexBatchId}}", + "cancel" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var j = pm.response.json();", + "pm.test('Vertex cancel batch: has id', function () { pm.expect(j.id).to.be.a('string').and.not.empty; });", + "pm.test('Vertex cancel batch: cancelling/cancelled', function () { pm.expect(['cancelling','cancelled']).to.include(j.status); });" + ] + } + } + ] + }, + { + "name": "[PREVIEW] Vertex: create batch RAW passthrough (native genai)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"displayName\": \"bifrost-harness-passthrough\",\n \"model\": \"publishers/google/models/gemini-2.5-flash\",\n \"inputConfig\": {\n \"instancesFormat\": \"jsonl\",\n \"gcsSource\": {\n \"uris\": [\n \"{{vertexBatchInputGcsUri}}\"\n ]\n }\n },\n \"outputConfig\": {\n \"predictionsFormat\": \"jsonl\",\n \"gcsDestination\": {\n \"outputUriPrefix\": \"gs://{{vertexGcsBucket}}/{{vertexGcsPrefix}}batch-output\"\n }\n }\n}", + "options": { + "raw": { + "language": "json" + } + } + }, + "url": { + "raw": "{{baseUrl}}/genai/v1/projects/{{vertexProject}}/locations/{{vertexLocation}}/batchPredictionJobs", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1", + "projects", + "{{vertexProject}}", + "locations", + "{{vertexLocation}}", + "batchPredictionJobs" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Sends a verbatim Vertex BatchPredictionJob body; Bifrost passes it through (UseRawRequestBody)", + "// and returns the native Vertex job resource. Proves the raw-passthrough path end to end.", + "var j = pm.response.json();", + "pm.test('Vertex RAW passthrough: name is a batchPredictionJobs resource', function () { pm.expect(j.name).to.be.a('string'); pm.expect(j.name).to.include('batchPredictionJobs'); });", + "pm.test('Vertex RAW passthrough: has JOB_STATE_ state', function () { pm.expect(j.state || '').to.match(/^JOB_STATE_/); });", + "var parts = (j.name || '').split('/'); pm.collectionVariables.set('vertexRawBatchId', parts[parts.length - 1]);" + ] + } + } + ] + }, + { + "name": "[PREVIEW] Vertex: delete batch RAW passthrough (native genai, cleanup)", + "request": { + "method": "DELETE", + "header": [], + "url": { + "raw": "{{baseUrl}}/genai/v1/projects/{{vertexProject}}/locations/{{vertexLocation}}/batchPredictionJobs/{{vertexRawBatchId}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1", + "projects", + "{{vertexProject}}", + "locations", + "{{vertexLocation}}", + "batchPredictionJobs", + "{{vertexRawBatchId}}" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('Vertex RAW passthrough cleanup: 2xx', function () { pm.expect(pm.response.code).to.be.within(200, 299); });" + ] + } + } + ] + }, + { + "name": "[PREVIEW] Vertex: delete batch input file (/openai, cleanup)", + "request": { + "method": "DELETE", + "header": [], + "url": { + "raw": "{{baseUrl}}/openai/v1/files/{{vertexBatchInputFileId}}?provider=vertex", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "files", + "{{vertexBatchInputFileId}}" + ], + "query": [ + { + "key": "provider", + "value": "vertex" + } + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "var j = pm.response.json();", + "pm.test('Vertex batch input cleanup: deleted', function () { pm.expect(j.deleted).to.eql(true); });" + ] + } + } + ] + } + ] + }, + { + "name": "12. Backlog Coverage (auto-added missing cases)", + "description": "Comprehensive coverage of features sourced from each provider's docs. Organized by provider. Many entries will fail in environments without the corresponding model/feature provisioned - that's expected; they exist to surface gaps via the failure report.", + "item": [ + { + "name": "OpenAI Backlog", + "item": [ + { + "name": "OpenAI: tool_choice specific function", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a color\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"pick_color\",\"parameters\":{\"type\":\"object\",\"properties\":{\"hex\":{\"type\":\"string\"}}}}}],\n \"tool_choice\": {\"type\":\"function\",\"function\":{\"name\":\"pick_color\"}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "OpenAI: parallel_tool_calls=false", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"What's the weather in NYC and SF?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}}],\n \"parallel_tool_calls\": false\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "OpenAI: response_format json_object", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"system\",\"content\":\"Output JSON only.\"},{\"role\":\"user\",\"content\":\"Tokyo population\"}],\n \"response_format\": {\"type\":\"json_object\"}\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "OpenAI: logprobs + top_logprobs", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"logprobs\": true,\n \"top_logprobs\": 5\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "OpenAI: seed for deterministic output", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a number\"}],\n \"seed\": 12345\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "OpenAI: stop sequences", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four, five\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "OpenAI: stream_options include_usage", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"stream\": true,\n \"stream_options\": {\"include_usage\": true}\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "OpenAI: predicted outputs", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Echo: hello world\"}],\n \"prediction\": {\"type\":\"content\",\"content\":\"hello world\"}\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "OpenAI: service_tier auto", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"service_tier\": \"auto\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "OpenAI: store + metadata", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"store\": true,\n \"metadata\": {\"harness\": \"backlog\"}\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "OpenAI Responses: reasoning summary", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"o3-mini\",\n \"input\": \"What's 17*23?\",\n \"reasoning\": {\"summary\": \"auto\"}\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "OpenAI Responses streaming: summary_index + obfuscation preserved", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code >= 400) { return; }", + "pm.test('summary_index and obfuscation survive stream', function () {", + " var body = pm.response.text() || '';", + " pm.expect(body, 'expected obfuscation in SSE body').to.include('\"obfuscation\"');", + " // reasoning.summary is model-discretionary: o3-mini may emit no summary on simple prompts.", + " // Only assert summary_index preservation when a reasoning summary was actually emitted.", + " if (body.indexOf('response.reasoning_summary') !== -1) {", + " pm.expect(body, 'reasoning summary emitted but summary_index stripped').to.include('\"summary_index\"');", + " }", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"o3-mini\",\n \"input\": \"Prove that the square root of 2 is irrational, step by step.\",\n \"reasoning\": {\"summary\": \"detailed\"},\n \"stream\": true,\n \"stream_options\": {\"include_obfuscation\": true}\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "OpenAI Responses streaming: assistant phase preserved (gpt-5.3-codex)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code >= 400) { return; }", + "pm.test('phase appears on assistant message items', function () {", + " var body = pm.response.text() || '';", + " var hasPhase = body.indexOf('\"phase\":\"final_answer\"') !== -1 || body.indexOf('\"phase\":\"commentary\"') !== -1;", + " pm.expect(hasPhase, 'no phase field in SSE body. First 200 chars: ' + body.slice(0,200)).to.be.true;", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-5.3-codex\",\n \"input\": \"Solve 2+2 and explain your steps briefly.\",\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "OpenAI Responses: assistant phase input round-trip (gpt-5.3-codex)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code >= 400) {", + " pm.test('phase field accepted on input message', function () {", + " var body = (pm.response.text() || '').toLowerCase();", + " pm.expect(body, 'unexpected rejection of phase field: ' + body.slice(0,200)).to.not.include('unknown field \"phase\"');", + " pm.expect(body).to.not.include('unexpected field \"phase\"');", + " });", + " return;", + "}", + "pm.test('response returned output items', function () {", + " var body = pm.response.text() || '';", + " pm.expect(body).to.include('\"output\"');", + "});" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-5.3-codex\",\n \"input\": [\n {\"role\":\"user\",\"content\":\"What's 2+2?\"},\n {\"role\":\"assistant\",\"phase\":\"final_answer\",\"content\":\"4\"},\n {\"role\":\"user\",\"content\":\"Now what's 3+3?\"}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "OpenAI Responses: background mode", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"input\": \"Hi\",\n \"background\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "OpenAI Responses: truncation auto", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"input\": \"Hi\",\n \"truncation\": \"auto\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "OpenAI Responses: include array", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"input\": \"Hi\",\n \"include\": [\"message.input_image.image_url\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "OpenAI Responses: custom tool", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"input\": \"Send Slack message\",\n \"tools\": [{\"type\":\"function\",\"name\":\"send_slack\",\"parameters\":{\"type\":\"object\",\"properties\":{\"channel\":{\"type\":\"string\"},\"message\":{\"type\":\"string\"}}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "OpenAI Responses: token counting", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"input\": \"Count tokens for me\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses/input_tokens", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses", + "input_tokens" + ] + } + } + } + ] + }, + { + "name": "Anthropic Backlog", + "item": [ + { + "name": "Anthropic: prompt caching 1h TTL", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "extended-cache-ttl-2025-04-11" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"long context\",\"cache_control\":{\"type\":\"ephemeral\",\"ttl\":\"1h\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: web_fetch tool", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Fetch https://example.com and summarize\"}],\n \"tools\": [{\"type\":\"web_fetch_20250910\",\"name\":\"web_fetch\",\"max_uses\":2}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: memory tool", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Remember my name is Akshay\"}],\n \"tools\": [{\"type\":\"memory_20250818\",\"name\":\"memory\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: tool_search BM25", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"List tools matching 'weather'\"}],\n \"tools\": [{\"type\":\"tool_search_tool_bm25\",\"name\":\"tool_search\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: tool_search regex", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Find tools matching get_.*\"}],\n \"tools\": [{\"type\":\"tool_search_tool_regex\",\"name\":\"tool_search\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: code_execution v2 (20250825)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50! using Python\"}],\n \"tools\": [{\"type\":\"code_execution_20250825\",\"name\":\"code_execution\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: code_execution programmatic (20260120)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Plot sin(x) and tell me the period\"}],\n \"tools\": [{\"type\":\"code_execution_20260120\",\"name\":\"code_execution\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: PDF input", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: stop sequences", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop_sequences\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: service_tier auto", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"service_tier\": \"auto\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: output_config effort high", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-6\",\n \"max_tokens\": 1024,\n \"output_config\": {\"effort\": \"high\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve x^2 - 5x + 6 = 0\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: output_config format json_schema", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"output_config\": {\"format\": {\"type\":\"json_schema\",\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}},\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a city\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: tool defer_loading + advanced beta", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "advanced-tool-use-2025-09-15" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"tools\": [{\"name\":\"slow_tool\",\"input_schema\":{\"type\":\"object\"},\"defer_loading\":true},{\"name\":\"fast_tool\",\"input_schema\":{\"type\":\"object\"},\"defer_loading\":false}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Call slow_tool\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: tool input_examples + beta", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "tool-examples-2025-10-29" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 1024,\n \"tools\": [{\"name\":\"f\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"x\":{\"type\":\"number\"}}},\"input_examples\":[{\"input\":{\"x\":1},\"description\":\"basic\"}]}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Call f\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: strict tool input", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"tools\": [{\"name\":\"strict_fn\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"a\":{\"type\":\"string\"}},\"required\":[\"a\"],\"additionalProperties\":false},\"strict\":true}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Call strict_fn\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: token counting", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"How many tokens?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages/count_tokens", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages", + "count_tokens" + ] + } + } + }, + { + "name": "Anthropic: list models", + "request": { + "method": "GET", + "header": [ + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "url": { + "raw": "{{baseUrl}}/anthropic/v1/models", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "models" + ] + } + } + }, + { + "name": "Anthropic: list batches", + "request": { + "method": "GET", + "header": [ + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages/batches", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages", + "batches" + ] + } + } + }, + { + "name": "Anthropic: list files", + "request": { + "method": "GET", + "header": [ + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "files-api-2025-04-14" + } + ], + "url": { + "raw": "{{baseUrl}}/anthropic/v1/files", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "files" + ] + } + } + }, + { + "name": "Anthropic: file upload with content_type override", + "request": { + "method": "POST", + "header": [ + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "files-api-2025-04-14" + } + ], + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.txt", + "type": "file" + }, + { + "key": "purpose", + "value": "user_data", + "type": "text" + }, + { + "key": "content_type", + "value": "text/markdown", + "type": "text" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/files", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "files" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('status is 200', function () { pm.response.to.have.status(200); });", + "var j = pm.response.json();", + "pm.test('has file id', function () { pm.expect(j.id).to.be.a('string').and.not.empty; });", + "pm.collectionVariables.set('anthropicUploadFileIdA', j.id);" + ] + } + } + ] + }, + { + "name": "Anthropic: file upload content_type inferred from file part", + "request": { + "method": "POST", + "header": [ + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "files-api-2025-04-14" + } + ], + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "file", + "src": "tests/e2e/api/fixtures/sample.txt", + "type": "file" + }, + { + "key": "purpose", + "value": "user_data", + "type": "text" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/files", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "files" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('status is 200', function () { pm.response.to.have.status(200); });", + "var j = pm.response.json();", + "pm.test('has file id', function () { pm.expect(j.id).to.be.a('string').and.not.empty; });", + "pm.collectionVariables.set('anthropicUploadFileIdB', j.id);" + ] + } + } + ] + }, + { + "name": "[PREVIEW] Anthropic: document file_id source (auto files-api beta header)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"file\",\"file_id\":\"{{anthropicUploadFileIdB}}\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "[PREVIEW] Anthropic Responses: input_file by file_id (native /v1/responses)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"input\": [{\"role\":\"user\",\"content\":[{\"type\":\"input_text\",\"text\":\"Summarize\"},{\"type\":\"input_file\",\"file_id\":\"{{anthropicUploadFileIdB}}\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "responses" + ] + } + } + }, + { + "name": "Anthropic: delete content_type test file A (cleanup)", + "request": { + "method": "DELETE", + "header": [ + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "files-api-2025-04-14" + } + ], + "url": { + "raw": "{{baseUrl}}/anthropic/v1/files/{{anthropicUploadFileIdA}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "files", + "{{anthropicUploadFileIdA}}" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('status is 200', function () { pm.response.to.have.status(200); });", + "var j = pm.response.json();", + "pm.test('Anthropic delete: file_deleted', function () { pm.expect(j.type).to.eql('file_deleted'); });", + "pm.test('Anthropic delete: id matches', function () { pm.expect(j.id).to.eql(pm.collectionVariables.get('anthropicUploadFileIdA')); });" + ] + } + } + ] + }, + { + "name": "Anthropic: delete content_type test file B (cleanup)", + "request": { + "method": "DELETE", + "header": [ + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "files-api-2025-04-14" + } + ], + "url": { + "raw": "{{baseUrl}}/anthropic/v1/files/{{anthropicUploadFileIdB}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "files", + "{{anthropicUploadFileIdB}}" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "pm.test('status is 200', function () { pm.response.to.have.status(200); });", + "var j = pm.response.json();", + "pm.test('Anthropic delete: file_deleted', function () { pm.expect(j.type).to.eql('file_deleted'); });", + "pm.test('Anthropic delete: id matches', function () { pm.expect(j.id).to.eql(pm.collectionVariables.get('anthropicUploadFileIdB')); });" + ] + } + } + ] + } + ] + }, + { + "name": "Anthropic Beta Headers", + "item": [ + { + "name": "Beta: token-efficient-tools", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "token-efficient-tools-2025-02-19" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"tools\": [{\"name\":\"f\",\"input_schema\":{\"type\":\"object\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Beta: fine-grained-tool-streaming", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "fine-grained-tool-streaming-2025-05-14" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"stream\": true,\n \"tools\": [{\"name\":\"f\",\"input_schema\":{\"type\":\"object\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "[PREVIEW] Beta: fast-mode (Opus 4.6)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "fast-mode-2026-02-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-6\",\n \"max_tokens\": 800,\n \"speed\": \"fast\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "[PREVIEW] Beta: fast-mode (Opus 4.7)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "fast-mode-2026-02-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 800,\n \"speed\": \"fast\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "[PREVIEW] Beta: fast-mode (Opus 4.8)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "fast-mode-2026-02-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 800,\n \"speed\": \"fast\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Beta: context-1m", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "context-1m-2025-09-15" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-6\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Beta: interleaved-thinking", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "interleaved-thinking-2025-05-14" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Beta: skills bundle", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "skills-2025-10-29" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Use skills\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Beta: redact-thinking", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "redact-thinking-2025-09-15" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Reasoned answer\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Beta: compaction", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "compact-2025-09-15" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-6\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Long convo\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Mid-conv system message (Opus 4.8) — system ends array", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 512,\n \"system\": [{\"type\":\"text\",\"text\":\"You are a helpful assistant.\"}],\n \"messages\": [\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Hello\"}]},\n {\"role\":\"assistant\",\"content\":[{\"type\":\"text\",\"text\":\"Hi!\"}]},\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"How are you?\"}]},\n {\"role\":\"system\",\"content\":[{\"type\":\"text\",\"text\":\"Respond only in one word.\"}]}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Mid-conv system message (Opus 4.8) — system mid-history (before assistant)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 512,\n \"system\": [{\"type\":\"text\",\"text\":\"You are a helpful assistant.\"}],\n \"messages\": [\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Hello\"}]},\n {\"role\":\"system\",\"content\":[{\"type\":\"text\",\"text\":\"From now on be very concise.\"}]},\n {\"role\":\"assistant\",\"content\":[{\"type\":\"text\",\"text\":\"Hi!\"}]},\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"How are you?\"}]}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + } + ] + }, + { + "name": "Bedrock Backlog", + "item": [ + { + "name": "Bedrock Converse: streaming", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Count to 5\"}]}],\n \"inferenceConfig\": {\"maxTokens\": 256}\n}" + }, + "url": { + "raw": "{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse-stream", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "{{bedrockModel}}", + "converse-stream" + ] + } + } + }, + { + "name": "Bedrock Converse: stop sequences", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Count: one, two, three, four in lowercase\"}]}],\n \"inferenceConfig\": {\"maxTokens\": 256, \"stopSequences\": [\"three\"]}\n}" + }, + "url": { + "raw": "{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "{{bedrockModel}}", + "converse" + ] + } + } + }, + { + "name": "Bedrock Converse: tool choice forced", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Pick a number\"}]}],\n \"toolConfig\": {\"tools\":[{\"toolSpec\":{\"name\":\"pick\",\"inputSchema\":{\"json\":{\"type\":\"object\",\"properties\":{\"n\":{\"type\":\"number\"}}}}}}],\"toolChoice\":{\"tool\":{\"name\":\"pick\"}}},\n \"inferenceConfig\": {\"maxTokens\":256}\n}" + }, + "url": { + "raw": "{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "{{bedrockModel}}", + "converse" + ] + } + } + }, + { + "name": "Bedrock Converse: performance config optimized", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hi\"}]}],\n \"performanceConfig\": {\"latency\":\"optimized\"},\n \"inferenceConfig\": {\"maxTokens\":256}\n}" + }, + "url": { + "raw": "{{baseUrl}}/bedrock/model/us.amazon.nova-pro-v1:0/converse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "us.amazon.nova-pro-v1:0", + "converse" + ] + } + } + }, + { + "name": "Bedrock Converse: request metadata", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hi\"}]}],\n \"requestMetadata\": {\"session\":\"harness\"},\n \"inferenceConfig\": {\"maxTokens\":256}\n}" + }, + "url": { + "raw": "{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "{{bedrockModel}}", + "converse" + ] + } + } + }, + { + "name": "Bedrock InvokeModel: direct Anthropic shape", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"anthropic_version\": \"bedrock-2023-05-31\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/bedrock/model/{{bedrockModel}}/invoke", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "{{bedrockModel}}", + "invoke" + ] + } + } + } + ] + }, + { + "name": "Gemini Backlog", + "item": [ + { + "name": "Gemini: tool config functionCallingConfig ANY", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Call get_weather\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"get_weather\",\"parameters\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"}}}}]}],\n \"toolConfig\": {\"functionCallingConfig\":{\"mode\":\"ANY\",\"allowedFunctionNames\":[\"get_weather\"]}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Gemini: stop sequences", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Count: one, two, three, four in lowercase\"}]}],\n \"generationConfig\": {\"stopSequences\":[\"three\"]}\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Gemini: temperature + topP + topK", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Pick a number\"}]}],\n \"generationConfig\": {\"temperature\":0.7,\"topP\":0.9,\"topK\":40}\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Gemini: response logprobs", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Hi\"}]}],\n \"generationConfig\": {\"responseLogprobs\":true,\"logprobs\":3}\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Gemini: presence + frequency penalty", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Tell a story\"}]}],\n \"generationConfig\": {\"presencePenalty\":0.5,\"frequencyPenalty\":0.5}\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "[PREVIEW] Gemini: PDF input", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Summarize this PDF\"},{\"fileData\":{\"mimeType\":\"application/pdf\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/data/A17_FlightPlan.pdf\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Gemini: YouTube URL input", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Summarize this video\"},{\"fileData\":{\"fileUri\":\"https://www.youtube.com/watch?v=jNQXAC9IVRw\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Gemini: URL context tool", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Summarize https://anthropic.com/news\"}]}],\n \"tools\": [{\"urlContext\":{}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Gemini: count tokens", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"How many tokens?\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:countTokens", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:countTokens" + ] + } + } + }, + { + "name": "Gemini: list models", + "request": { + "method": "GET", + "header": [ + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models" + ] + } + } + } + ] + }, + { + "name": "Vertex Backlog", + "item": [ + { + "name": "Vertex: anthropic_version in body", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"anthropic_version\": \"vertex-2023-10-16\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi via Vertex\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex: streaming Anthropic", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "[PREVIEW] Vertex Model Garden: Llama (publishers/meta/...)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/meta/llama-4-maverick-17b-128e-instruct-maas\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "[PREVIEW] Vertex Model Garden: Mistral", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/publishers/mistralai/models/mistral-large-2411\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Azure Backlog", + "item": [ + { + "name": "Azure: tools (function calling)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":\"What's the weather?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } + ] + } + } + }, + { + "name": "Azure: streaming", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } + ] + } + } + }, + { + "name": "Azure: structured output json_schema", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country for Paris\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"}},\"required\":[\"city\",\"country\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } + ] + } + } + }, + { + "name": "Azure: vision (image_url)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } + ] + } + } + }, + { + "name": "Azure: system + multi-turn", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Arrr!\"},{\"role\":\"user\",\"content\":\"Tell a joke\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } + ] + } + } + }, + { + "name": "Azure On Your Data: azure_search", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":\"What's in the docs?\"}],\n \"data_sources\": [{\"type\":\"azure_search\",\"parameters\":{\"endpoint\":\"https://placeholder.search.windows.net\",\"index_name\":\"placeholder\",\"authentication\":{\"type\":\"api_key\",\"key\":\"placeholder\"}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } + ] + } + } + } + ] + }, + { + "name": "Cross-Provider Backlog", + "item": [ + { + "name": "Cross-cut: code execution Anthropic", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50!\"}],\n \"tools\": [{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: code execution Gemini", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50!\"}],\n \"tools\": [{\"type\":\"code_execution\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: extended thinking via cross-model", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Plan a trip\"}],\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"max_tokens\": 4096\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: prompt caching via cross-model", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"system\",\"content\":[{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}]},{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: stop sequences (OpenAI)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked into content').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter'], 'unexpected finish_reason: ' + fr).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: stop sequences (Anthropic)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked into content').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter'], 'unexpected finish_reason: ' + fr).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: stop sequences (Gemini)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked into content').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter'], 'unexpected finish_reason: ' + fr).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: tool_choice forced (OpenAI)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"f\",\"parameters\":{\"type\":\"object\"}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: tool_choice forced (Bedrock)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"f\",\"parameters\":{\"type\":\"object\"}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: tool_choice forced (Bedrock)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"f\",\"parameters\":{\"type\":\"object\"}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: structured output Anthropic", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a city\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: structured output Bedrock", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a city\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: structured output Bedrock", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a city\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vision Bedrock", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vision Vertex (Gemini)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: web search Bedrock (Anthropic)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: failover scenario (rate-limit-trigger)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "X-Bifrost-Fallback-Models", + "value": "openai/gpt-4o-mini,anthropic/claude-haiku-4-5" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: virtual key auth (X-Bifrost-VK)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "X-Bifrost-VK", + "value": "vk-test" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: sampling-params silently dropped for Opus 4.7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"temperature\": 0.7,\n \"top_p\": 0.9\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: sampling-params silently dropped for Opus 4.8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"temperature\": 0.7,\n \"top_p\": 0.9\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Passthrough Backlog", + "item": [ + { + "name": "Passthrough OpenAI: web_search", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"input\": \"Latest AI news\",\n \"tools\": [{\"type\":\"web_search_preview\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai_passthrough/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai_passthrough", + "v1", + "responses" + ] + } + } + }, + { + "name": "Passthrough OpenAI: code_interpreter", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"input\": \"Plot sin(x)\",\n \"tools\": [{\"type\":\"code_interpreter\",\"container\":{\"type\":\"auto\"}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai_passthrough/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai_passthrough", + "v1", + "responses" + ] + } + } + }, + { + "name": "Passthrough Anthropic: computer_use", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "computer-use-2025-11-24" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"tools\": [{\"type\":\"computer_20251124\",\"name\":\"computer\",\"display_width_px\":1024,\"display_height_px\":768},{\"type\":\"text_editor_20250728\",\"name\":\"str_replace_based_edit_tool\"},{\"type\":\"bash_20250124\",\"name\":\"bash\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic_passthrough/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic_passthrough", + "v1", + "messages" + ] + } + } + }, + { + "name": "Passthrough Anthropic: extended thinking", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic_passthrough/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic_passthrough", + "v1", + "messages" + ] + } + } + }, + { + "name": "Passthrough Anthropic: prompt caching", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic_passthrough/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic_passthrough", + "v1", + "messages" + ] + } + } + }, + { + "name": "Passthrough Anthropic: web_search", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic_passthrough/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic_passthrough", + "v1", + "messages" + ] + } + } + }, + { + "name": "Passthrough GenAI: googleSearch", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Latest news\"}]}],\n \"tools\": [{\"googleSearch\":{}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai_passthrough/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai_passthrough", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Passthrough GenAI: codeExecution", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Compute fib(20)\"}]}],\n \"tools\": [{\"codeExecution\":{}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai_passthrough/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai_passthrough", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Passthrough Azure: tools", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":\"What's the weather?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/azure_passthrough/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "azure_passthrough", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } + ] + } + } + }, + { + "name": "Passthrough OpenAI: streaming chat", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai_passthrough/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai_passthrough", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Passthrough OpenAI: vision", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai_passthrough/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai_passthrough", + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Vertex Backlog Round 2 (Gemini-on-Vertex variants for system/multi-turn/sampling/tools)", + "item": [ + { + "name": "Vertex Gemini: system instruction", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"systemInstruction\": {\"parts\":[{\"text\":\"You are a chef\"}]},\n \"contents\": [{\"parts\":[{\"text\":\"I have eggs\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:generateContent" + ] + } + } + }, + { + "name": "Vertex Gemini: multi-turn history", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"role\":\"user\",\"parts\":[{\"text\":\"Hi\"}]},{\"role\":\"model\",\"parts\":[{\"text\":\"Hello\"}]},{\"role\":\"user\",\"parts\":[{\"text\":\"How are you?\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:generateContent" + ] + } + } + }, + { + "name": "Vertex Gemini: stop sequences", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Count: one, two, three, four in lowercase\"}]}],\n \"generationConfig\": {\"stopSequences\":[\"three\"]}\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:generateContent" + ] + } + } + }, + { + "name": "Vertex Gemini: sampling params", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Hi\"}]}],\n \"generationConfig\": {\"temperature\":0.7,\"topP\":0.9,\"topK\":40}\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:generateContent" + ] + } + } + }, + { + "name": "Vertex Gemini: tool choice forced (ANY)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Hi\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"f\",\"parameters\":{\"type\":\"OBJECT\"}}]}],\n \"toolConfig\": {\"functionCallingConfig\":{\"mode\":\"ANY\"}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:generateContent" + ] + } + } + }, + { + "name": "Vertex Gemini: streaming", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Count 1-5\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:streamGenerateContent?alt=sse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:streamGenerateContent" + ], + "query": [ + { + "key": "alt", + "value": "sse" + } + ] + } + } + }, + { + "name": "Vertex Gemini: vision", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:generateContent" + ] + } + } + }, + { + "name": "Vertex Gemini: code execution", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Compute fib(20)\"}]}],\n \"tools\": [{\"codeExecution\":{}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:generateContent" + ] + } + } + }, + { + "name": "Vertex Gemini: thinking budget", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Solve 17*23\"}]}],\n \"generationConfig\": {\"thinkingConfig\":{\"thinkingBudget\":4000}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:generateContent" + ] + } + } + }, + { + "name": "Vertex Gemini: structured output json_schema (via /v1/chat)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a city\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: extended thinking", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: web search", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: prompt caching", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "[PREVIEW] Vertex Claude: computer use", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"tools\": [{\"type\":\"computer_20250124\",\"name\":\"computer\",\"display_width_px\":1024,\"display_height_px\":768},{\"type\":\"text_editor_20250728\",\"name\":\"str_replace_based_edit_tool\"},{\"type\":\"bash_20250124\",\"name\":\"bash\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Bedrock Backlog Round 2 (Anthropic features via Bedrock)", + "item": [ + { + "name": "Bedrock: multi-turn", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hi\"}]},{\"role\":\"assistant\",\"content\":[{\"text\":\"Hello\"}]},{\"role\":\"user\",\"content\":[{\"text\":\"How are you?\"}]}],\n \"inferenceConfig\": {\"maxTokens\":256}\n}" + }, + "url": { + "raw": "{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "{{bedrockModel}}", + "converse" + ] + } + } + }, + { + "name": "Bedrock Converse: vision (image content)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Describe\"},{\"image\":{\"format\":\"jpeg\",\"source\":{\"bytes\":\"REPLACE_WITH_BASE64\"}}}]}],\n \"inferenceConfig\": {\"maxTokens\":512}\n}" + }, + "url": { + "raw": "{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "{{bedrockModel}}", + "converse" + ] + } + } + }, + { + "name": "Bedrock via /v1: extended thinking (Sonnet 4.6)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock via /v1: adaptive thinking (Opus 4.7)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock via /v1: adaptive thinking (Opus 4.8)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock via /v1: mid-conv system fallback (Opus 4.8 — merged to top-level system)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"},\n {\"role\":\"system\",\"content\":\"Respond only in one word.\"}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock via /v1: web search (Anthropic)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock via /v1: code execution (Anthropic)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50!\"}],\n \"tools\": [{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock via /v1: prompt caching (Anthropic)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock via /v1: computer use", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"tools\": [{\"type\":\"computer_20251124\",\"name\":\"computer\",\"display_width_px\":1024,\"display_height_px\":768},{\"type\":\"text_editor_20250728\",\"name\":\"str_replace_based_edit_tool\"},{\"type\":\"bash_20250124\",\"name\":\"bash\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Screenshot\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock via /v1: PDF input (Anthropic)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock Converse: sampling params", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hi\"}]}],\n \"inferenceConfig\": {\"maxTokens\":256, \"temperature\":0.7}\n}" + }, + "url": { + "raw": "{{baseUrl}}/bedrock/model/global.anthropic.claude-sonnet-4-6/converse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "global.anthropic.claude-sonnet-4-6", + "converse" + ] + } + } + } + ] + }, + { + "name": "Bedrock Mantle Backlog Round 2 (Anthropic features via Bedrock)", + "item": [ + { + "name": "Bedrock via /v1: extended thinking (Sonnet 4.6)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock via /v1: adaptive thinking (Opus 4.7)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock via /v1: adaptive thinking (Opus 4.8)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock via /v1: mid-conv system fallback (Opus 4.8 — merged to top-level system)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"},\n {\"role\":\"system\",\"content\":\"Respond only in one word.\"}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock via /v1: prompt caching (Anthropic)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "OpenAI/Azure Backlog Round 2", + "item": [ + { + "name": "OpenAI: sampling params", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"temperature\": 0.7,\n \"top_p\": 0.9,\n \"frequency_penalty\": 0.5,\n \"presence_penalty\": 0.3\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "[PREVIEW] OpenAI Responses: MCP tool", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"input\": \"Use mcp\",\n \"tools\": [{\"type\":\"mcp\",\"server_label\":\"test\",\"server_url\":\"https://example.com/mcp\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "[PREVIEW] OpenAI Responses: computer_use_preview", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"computer-use-preview\",\n \"input\": \"Take a screenshot\",\n \"truncation\": \"auto\",\n \"tools\": [{\"type\":\"computer_use_preview\",\"display_width\":1024,\"display_height\":768,\"environment\":\"browser\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "[PREVIEW] OpenAI Responses: PDF input via input_file", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"input\": [{\"role\":\"user\",\"content\":[{\"type\":\"input_text\",\"text\":\"Summarize\"},{\"type\":\"input_file\",\"file_id\":\"file_REPLACE_ME\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "[PREVIEW] OpenAI: audio input (gpt-4o-audio)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-audio-preview\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe audio\"},{\"type\":\"input_audio\",\"input_audio\":{\"data\":\"REPLACE_BASE64\",\"format\":\"wav\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "OpenAI: prompt caching marker (system reuse)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a helpful assistant. Long instructions for caching benefit. Long instructions for caching benefit. Long instructions for caching benefit.\"},{\"role\":\"user\",\"content\":\"Hi\"}],\n \"prompt_cache_key\": \"harness-cache-1\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "OpenAI: list batches", + "request": { + "method": "GET", + "header": [ + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "url": { + "raw": "{{baseUrl}}/openai/v1/batches", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "batches" + ] + } + } + }, + { + "name": "OpenAI: list files", + "request": { + "method": "GET", + "header": [ + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "url": { + "raw": "{{baseUrl}}/openai/v1/files", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "files" + ] + } + } + }, + { + "name": "OpenAI: list models", + "request": { + "method": "GET", + "header": [ + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "url": { + "raw": "{{baseUrl}}/openai/v1/models", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "models" + ] + } + } + }, + { + "name": "Azure: stop sequences", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } + ] + } + } + }, + { + "name": "Azure: sampling params", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"temperature\": 0.7,\n \"top_p\": 0.9\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } + ] + } + } + }, + { + "name": "Azure: tool_choice forced", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"f\",\"parameters\":{\"type\":\"object\"}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } + ] + } + } + }, + { + "name": "Azure: parallel_tool_calls=false", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"f\",\"parameters\":{\"type\":\"object\"}}}],\n \"parallel_tool_calls\": false\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } + ] + } + } + }, + { + "name": "[PREVIEW] Azure: reasoning_effort (o3 deployment)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"reasoning_effort\": \"high\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/o3/chat/completions?api-version={{azureApiVersion}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "o3", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } + ] + } + } + } + ] + }, + { + "name": "Anthropic + Gemini Backlog Round 2", + "item": [ + { + "name": "Anthropic: explicit multi-turn (3 turns)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: parallel tool calls disabled", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in NYC and SF?\"}],\n \"tools\": [{\"name\":\"get_weather\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}],\n \"tool_choice\": {\"type\":\"auto\",\"disable_parallel_tool_use\":true}\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "[PREVIEW] Anthropic: MCP toolset", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "mcp-client-2025-09-15" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"mcp_servers\": [{\"type\":\"url\",\"url\":\"https://example.com/mcp\",\"name\":\"test\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Use MCP\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: citations on document", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"text\",\"media_type\":\"text/plain\",\"data\":\"The sky is blue. Grass is green.\"},\"citations\":{\"enabled\":true}},{\"type\":\"text\",\"text\":\"What color is the sky?\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: eager input streaming + beta", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "eager-input-streaming-2025-10-29" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"stream\": true,\n \"tools\": [{\"name\":\"f\",\"input_schema\":{\"type\":\"object\"},\"eager_input_streaming\":true}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: allowed_callers + advanced beta", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "advanced-tool-use-2025-09-15" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 800,\n \"tools\": [{\"name\":\"f\",\"input_schema\":{\"type\":\"object\"},\"allowed_callers\":[\"direct\"]}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Anthropic: skills/container", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "skills-2025-10-29" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"container\": {\"skills\":[{\"skill_id\":\"data-analysis\",\"type\":\"anthropic\"}]},\n \"messages\": [{\"role\":\"user\",\"content\":\"Analyze\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Gemini: parallel function calls", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Weather in NYC and SF?\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"get_weather\",\"parameters\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"}}}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Gemini: structured output via /v1/chat (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a city\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "[PREVIEW] Gemini: cached content reference", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Use cache\"}]}],\n \"cachedContent\": \"cachedContents/REPLACE_ME\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Gemini: audio input (inline)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Transcribe\"},{\"inlineData\":{\"mimeType\":\"audio/wav\",\"data\":\"UklGRkQDAABXQVZFZm10IBAAAAABAAEAQB8AAIA+AAACABAAZGF0YSADAAAAAJUK6xPlGrIe3R5iG6oUgAv7AFj22eye5YHhAOEo5Jzql/MK/rcIXhLYGUQeHB9GHBgWTg3wAjv4cO645v7h0OBS4znp0fEW/NEGvhCxGLgdPB8OHXAXDg/jBCX6GPDs55niwOCZ4uznGPAl+uMEDg9wFw4dPB+4HbEYvhDRBhb80fE56VLj0OD+4bjmcO47+PACTg0YFkYcHB9EHtgZXhK3CAr+l/Oc6ijkAOGB4Z7l2exY9vsAgAuqFGIb3R6yHuUa6xOVCgAAa/UV7BvlTuEj4Z7kVuuA9AX/qAknE2Iafx4AH9gbZBVpDPYBSfei7SjmvOHk4Lrj6Omy8hD9xQeQEUgZAh4wH64cxxYvDuoDL/lC70/nSOLE4PLikOjy8B372wXoDxQYZx1AH2cdFBjoD9sFHfvy8JDo8uLE4EjiT+dC7y/56gMvDscWrhwwHwIeSBmQEcUHEP2y8ujpuuPk4LzhKOai7Un39gFpDGQV2BsAH38eYhonE6gJBf+A9FbrnuQj4U7hG+UV7Gv1AACVCusT5RqyHt0eYhuqFIAL+wBY9tnsnuWB4QDhKOSc6pfzCv63CF4S2BlEHhwfRhwYFk4N8AI7+HDuuOb+4dDgUuM56dHxFvzRBr4QsRi4HTwfDh1wFw4P4wQl+hjw7OeZ4sDgmeLs5xjwJfrjBA4PcBcOHTwfuB2xGL4Q0QYW/NHxOelS49Dg/uG45nDuO/jwAk4NGBZGHBwfRB7YGV4StwgK/pfznOoo5ADhgeGe5dnsWPb7AIALqhRiG90esh7lGusTlQoAAGv1Fewb5U7hI+Ge5FbrgPQF/6gJJxNiGn8eAB/YG2QVaQz2AUn3ou0o5rzh5OC64+jpsvIQ/cUHkBFIGQIeMB+uHMcWLw7qAy/5Qu9P50jixODy4pDo8vAd+9sF6A8UGGcdQB9nHRQY6A/bBR378vCQ6PLixOBI4k/nQu8v+eoDLw7HFq4cMB8CHkgZkBHFBxD9svLo6brj5OC84Sjmou1J9/YBaQxkFdgbAB9/HmIaJxOoCQX/gPRW657kI+FO4RvlFexr9Q==\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Gemini: list cached contents", + "request": { + "method": "GET", + "header": [ + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "url": { + "raw": "{{baseUrl}}/genai/v1beta/cachedContents", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "cachedContents" + ] + } + } + }, + { + "name": "Gemini: list files", + "request": { + "method": "GET", + "header": [ + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "url": { + "raw": "{{baseUrl}}/genai/v1beta/files", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "files" + ] + } + } + } + ] + }, + { + "name": "Bedrock Round 3 (every remaining gap)", + "item": [ + { + "name": "Bedrock: stop sequences via /v1 (Anthropic)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: parallel_tool_calls disabled", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"f\",\"parameters\":{\"type\":\"object\"}}}],\n \"parallel_tool_calls\": false\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: MCP toolset", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"mcp_servers\": [{\"type\":\"url\",\"url\":\"https://example.com/mcp\",\"name\":\"test\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Use MCP\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: web_search dynamic filtering", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"messages\": [{\"role\":\"user\",\"content\":\"Search\"}],\n \"tools\": [{\"type\":\"web_search_20260209\",\"name\":\"web_search\"},{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: web_search domain filter", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Search\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2,\"allowed_domains\":[\"arxiv.org\"]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: web_search user_location", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Local search\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"user_location\":{\"type\":\"approximate\",\"city\":\"SF\",\"country\":\"US\"}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: web_fetch tool", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Fetch URL\"}],\n \"tools\": [{\"type\":\"web_fetch_20250910\",\"name\":\"web_fetch\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: text_editor tool", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"tools\": [{\"type\":\"text_editor_20250728\",\"name\":\"str_replace_based_edit_tool\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Edit\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: bash tool", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"tools\": [{\"type\":\"bash_20250124\",\"name\":\"bash\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Run ls\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: memory tool", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"tools\": [{\"type\":\"memory_20250818\",\"name\":\"memory\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Remember\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: citations on document", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"text\",\"media_type\":\"text/plain\",\"data\":\"Sky is blue\"},\"citations\":{\"enabled\":true}},{\"type\":\"text\",\"text\":\"What color?\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: prompt caching 1h", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long\",\"cache_control\":{\"type\":\"ephemeral\",\"ttl\":\"1h\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: anthropic-beta header", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-beta", + "value": "prompt-caching-2024-07-31" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: interleaved thinking", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-beta", + "value": "interleaved-thinking-2025-05-14" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: context management", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-beta", + "value": "context-management-2025-09-15" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: service_tier auto", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"max_tokens\": 800,\n \"service_tier\": \"default\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: list invocation jobs (Batch)", + "request": { + "method": "GET", + "header": [], + "url": { + "raw": "{{baseUrl}}/bedrock/model-invocation-jobs", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model-invocation-jobs" + ] + } + } + } + ] + }, + { + "name": "Bedrock Mantle Round 3 (every remaining gap)", + "item": [ + { + "name": "Bedrock: stop sequences via /v1 (Anthropic)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: parallel_tool_calls disabled", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"f\",\"parameters\":{\"type\":\"object\"}}}],\n \"parallel_tool_calls\": false\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: text_editor tool", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 1024,\n \"tools\": [{\"type\":\"text_editor_20250728\",\"name\":\"str_replace_based_edit_tool\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Edit\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: bash tool", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 1024,\n \"tools\": [{\"type\":\"bash_20250124\",\"name\":\"bash\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Run ls\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: memory tool", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 1024,\n \"tools\": [{\"type\":\"memory_20250818\",\"name\":\"memory\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Remember\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: citations on document", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"text\",\"media_type\":\"text/plain\",\"data\":\"Sky is blue\"},\"citations\":{\"enabled\":true}},{\"type\":\"text\",\"text\":\"What color?\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: prompt caching 1h", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long\",\"cache_control\":{\"type\":\"ephemeral\",\"ttl\":\"1h\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: anthropic-beta header", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-beta", + "value": "prompt-caching-2024-07-31" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: interleaved thinking", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-beta", + "value": "interleaved-thinking-2025-05-14" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: context management", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-beta", + "value": "context-management-2025-09-15" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Bedrock: service_tier auto", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 800,\n \"service_tier\": \"default\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Vertex Round 3 (every remaining gap)", + "item": [ + { + "name": "Vertex Gemini: parallel function calls", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Weather NYC and SF\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"get_weather\",\"parameters\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"}}}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:generateContent" + ] + } + } + }, + { + "name": "Vertex Claude: tool_search", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Find tools\"}],\n \"tools\": [{\"type\":\"tool_search_tool_bm25\",\"name\":\"tool_search_tool_bm25\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: MCP toolset", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"mcp_servers\": [{\"type\":\"url\",\"url\":\"https://example.com/mcp\",\"name\":\"t\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Use MCP\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: web search dynamic", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"messages\": [{\"role\":\"user\",\"content\":\"Search\"}],\n \"tools\": [{\"type\":\"web_search_20260209\",\"name\":\"web_search\"},{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: web search domain", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Search\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"allowed_domains\":[\"arxiv.org\"]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: web search location", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Local\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"user_location\":{\"type\":\"approximate\",\"city\":\"SF\",\"country\":\"US\"}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: web_fetch", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Fetch URL\"}],\n \"tools\": [{\"type\":\"web_fetch_20250910\",\"name\":\"web_fetch\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: memory tool", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"tools\": [{\"type\":\"memory_20250818\",\"name\":\"memory\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Remember\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: PDF input", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Gemini: audio input", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Transcribe\"},{\"inlineData\":{\"mimeType\":\"audio/wav\",\"data\":\"UklGRkQDAABXQVZFZm10IBAAAAABAAEAQB8AAIA+AAACABAAZGF0YSADAAAAAJUK6xPlGrIe3R5iG6oUgAv7AFj22eye5YHhAOEo5Jzql/MK/rcIXhLYGUQeHB9GHBgWTg3wAjv4cO645v7h0OBS4znp0fEW/NEGvhCxGLgdPB8OHXAXDg/jBCX6GPDs55niwOCZ4uznGPAl+uMEDg9wFw4dPB+4HbEYvhDRBhb80fE56VLj0OD+4bjmcO47+PACTg0YFkYcHB9EHtgZXhK3CAr+l/Oc6ijkAOGB4Z7l2exY9vsAgAuqFGIb3R6yHuUa6xOVCgAAa/UV7BvlTuEj4Z7kVuuA9AX/qAknE2Iafx4AH9gbZBVpDPYBSfei7SjmvOHk4Lrj6Omy8hD9xQeQEUgZAh4wH64cxxYvDuoDL/lC70/nSOLE4PLikOjy8B372wXoDxQYZx1AH2cdFBjoD9sFHfvy8JDo8uLE4EjiT+dC7y/56gMvDscWrhwwHwIeSBmQEcUHEP2y8ujpuuPk4LzhKOai7Un39gFpDGQV2BsAH38eYhonE6gJBf+A9FbrnuQj4U7hG+UV7Gv1AACVCusT5RqyHt0eYhuqFIAL+wBY9tnsnuWB4QDhKOSc6pfzCv63CF4S2BlEHhwfRhwYFk4N8AI7+HDuuOb+4dDgUuM56dHxFvzRBr4QsRi4HTwfDh1wFw4P4wQl+hjw7OeZ4sDgmeLs5xjwJfrjBA4PcBcOHTwfuB2xGL4Q0QYW/NHxOelS49Dg/uG45nDuO/jwAk4NGBZGHBwfRB7YGV4StwgK/pfznOoo5ADhgeGe5dnsWPb7AIALqhRiG90esh7lGusTlQoAAGv1Fewb5U7hI+Ge5FbrgPQF/6gJJxNiGn8eAB/YG2QVaQz2AUn3ou0o5rzh5OC64+jpsvIQ/cUHkBFIGQIeMB+uHMcWLw7qAy/5Qu9P50jixODy4pDo8vAd+9sF6A8UGGcdQB9nHRQY6A/bBR378vCQ6PLixOBI4k/nQu8v+eoDLw7HFq4cMB8CHkgZkBHFBxD9svLo6brj5OC84Sjmou1J9/YBaQxkFdgbAB9/HmIaJxOoCQX/gPRW657kI+FO4RvlFexr9Q==\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:generateContent" + ] + } + } + }, + { + "name": "Vertex Claude: citations", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"text\",\"media_type\":\"text/plain\",\"data\":\"Sky blue\"},\"citations\":{\"enabled\":true}},{\"type\":\"text\",\"text\":\"Color?\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: extended thinking budget_tokens", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: prompt caching 1h", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long\",\"cache_control\":{\"type\":\"ephemeral\",\"ttl\":\"1h\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: anthropic-beta header", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-beta", + "value": "prompt-caching-2024-07-31" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: defer_loading", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"tools\": [{\"name\":\"f\",\"input_schema\":{\"type\":\"object\"},\"defer_loading\":true}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: allowed_callers", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"tools\": [{\"name\":\"f\",\"input_schema\":{\"type\":\"object\"},\"allowed_callers\":[\"a\"]}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: interleaved thinking", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-beta", + "value": "interleaved-thinking-2025-05-14" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: context management", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-beta", + "value": "context-management-2025-09-15" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Gemini: safety settings BLOCK_NONE", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Tell a story\"}]}],\n \"safetySettings\": [{\"category\":\"HARM_CATEGORY_DANGEROUS_CONTENT\",\"threshold\":\"BLOCK_NONE\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{vertexModel}}:generateContent" + ] + } + } + } + ] + }, + { + "name": "OpenAI/Anthropic/Gemini/Azure Round 3 (final gap closure)", + "item": [ + { + "name": "[PREVIEW] OpenAI: file_search w/ placeholder vector store", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"input\": \"Search KB\",\n \"tools\": [{\"type\":\"file_search\",\"vector_store_ids\":[\"vs_REPLACE_ME\"]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses" + ] + } + } + }, + { + "name": "OpenAI: token counting endpoint", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"input\": \"Count my tokens\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses/input_tokens", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses", + "input_tokens" + ] + } + } + }, + { + "name": "OpenAI: create batch", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"input_file_id\": \"file_REPLACE_ME\",\n \"endpoint\": \"/v1/chat/completions\",\n \"completion_window\": \"24h\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/batches", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "batches" + ] + } + } + }, + { + "name": "Anthropic: token counting endpoint", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"How many tokens in this message?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages/count_tokens", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages", + "count_tokens" + ] + } + } + }, + { + "name": "Anthropic: create batch", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"requests\": [{\"custom_id\":\"req-1\",\"params\":{\"model\":\"claude-haiku-4-5\",\"max_tokens\":256,\"messages\":[{\"role\":\"user\",\"content\":\"Hi\"}]}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages/batches", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages", + "batches" + ] + } + } + }, + { + "name": "Gemini: tool_choice forced via toolConfig", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Hi\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"f\",\"parameters\":{\"type\":\"OBJECT\"}}]}],\n \"toolConfig\": {\"functionCallingConfig\":{\"mode\":\"ANY\",\"allowedFunctionNames\":[\"f\"]}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "[PREVIEW] Gemini: prompt caching via cachedContents", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Use cache\"}]}],\n \"cachedContent\": \"cachedContents/PLACEHOLDER\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:generateContent" + ] + } + } + }, + { + "name": "Gemini: token counting", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"How many tokens?\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:countTokens", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "{{genaiModel}}:countTokens" + ] + } + } + }, + { + "name": "Azure: code_interpreter via Responses preview", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"input\": \"Plot sin(x)\",\n \"tools\": [{\"type\":\"code_interpreter\",\"container\":{\"type\":\"auto\"}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/responses?api-version=2025-04-01-preview", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "{{azureDeployment}}", + "responses" + ], + "query": [ + { + "key": "api-version", + "value": "2025-04-01-preview" + } + ] + } + } + }, + { + "name": "[PREVIEW] Azure: file_search via Responses preview", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"input\": \"Search\",\n \"tools\": [{\"type\":\"file_search\",\"vector_store_ids\":[\"vs_REPLACE_ME\"]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/responses?api-version=2025-04-01-preview", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "{{azureDeployment}}", + "responses" + ], + "query": [ + { + "key": "api-version", + "value": "2025-04-01-preview" + } + ] + } + } + }, + { + "name": "[PREVIEW] Azure: audio (gpt-4o-audio deployment)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"input_audio\",\"input_audio\":{\"data\":\"REPLACE_BASE64\",\"format\":\"wav\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/gpt-4o-audio-preview/chat/completions?api-version={{azureApiVersion}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "gpt-4o-audio-preview", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } + ] + } + } + }, + { + "name": "[PREVIEW] Azure: skills/container", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"input\": \"Use skills\",\n \"tools\": [{\"type\":\"code_interpreter\",\"container\":{\"type\":\"auto\"}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/responses?api-version=2025-04-01-preview", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "{{azureDeployment}}", + "responses" + ], + "query": [ + { + "key": "api-version", + "value": "2025-04-01-preview" + } + ] + } + } + }, + { + "name": "Azure: service_tier scale", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"service_tier\": \"flex\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } + ] + } + } + } + ] + }, + { + "name": "Cross-Cut Round 4 (Vertex Claude + Vertex Gemini + Azure cross-cut)", + "item": [ + { + "name": "Vertex Claude: structured output (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (CityInfo)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('name').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p.name.toLowerCase(), 'expected Paris').to.include('paris'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"system\",\"content\":\"Extract the city information from the user's message.\"},{\"role\":\"user\",\"content\":\"I visited Paris, France last summer.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"CityInfo\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"name\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"}},\"required\":[\"name\",\"country\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: function calling cross-cut", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: streaming", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: vision", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"What is in this image?\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: tool_choice forced", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"f\",\"parameters\":{\"type\":\"object\"}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: stop sequences", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: multi-turn cross-cut", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: system message cross-cut", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: web search via /v1/chat (sonnet)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: code execution via /v1/chat (opus)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50!\"}],\n \"tools\": [{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: PDF input via /v1/chat (sonnet)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: sampling-params dropped for Opus 4.7", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"temperature\": 0.7,\n \"top_p\": 0.9\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Claude: sampling-params dropped for Opus 4.8", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"temperature\": 0.7,\n \"top_p\": 0.9\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Gemini: function calling cross-cut", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Vertex Gemini: streaming cross-cut", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: web search Vertex Gemini (google_search)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"google_search\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: stop sequences (Bedrock)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked into content').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter'], 'unexpected finish_reason: ' + fr).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: stop sequences (Bedrock)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked into content').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter'], 'unexpected finish_reason: ' + fr).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: stop sequences (Vertex Gemini)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked into content').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter'], 'unexpected finish_reason: ' + fr).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: Azure basic chat", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: Azure tools", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: Azure structured output (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country for Paris\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"}},\"required\":[\"city\",\"country\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: Azure streaming", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: Azure vision", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Cross-Cut Round 5: Structured Output Matrix (response_format json_schema via /v1/chat across providers/models)", + "item": [ + { + "name": "Cross-cut: openai/gpt-5 (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-5-mini (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-4o (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-4.1 (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4.1\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/o3-mini (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/o3-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-opus-4-7 (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-sonnet-4-6 (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-opus-4-7 (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-opus-4-7 (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-haiku-4-5 (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-haiku-4-5 (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "[PREVIEW] Cross-cut: bedrock/nova-pro (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-pro-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock_mantle/anthropic.claude-opus-4-8 (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "[PREVIEW] Cross-cut: bedrock/nova-lite (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock_mantle/anthropic.claude-opus-4-8 (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-opus-4-7 (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/gemini-2.5-flash (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: gemini/gemini-2.5-pro (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Cross-Cut Round 6: Function Calling Matrix", + "item": [ + { + "name": "Cross-cut: openai/gpt-5 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-5-mini function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-4o function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-4.1 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4.1\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/o3-mini function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/o3-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-opus-4-7 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-sonnet-4-6 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-haiku-4-5 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "[PREVIEW] Cross-cut: bedrock/us.amazon.nova-pro-v1:0 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-pro-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock_mantle/anthropic.claude-opus-4-8 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-opus-4-7 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-sonnet-4-6 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/gemini-2.5-pro function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/gemini-2.5-flash function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: gemini/gemini-2.5-pro function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: gemini/gemini-2.5-flash function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: azure/{{azureDeployment}} function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Cross-Cut Round 7: Streaming Matrix", + "item": [ + { + "name": "Cross-cut: openai/gpt-5 streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-5-mini streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-4o streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-4.1 streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4.1\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/o3-mini streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/o3-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-opus-4-7 streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-sonnet-4-6 streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-haiku-4-5 streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0 streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0 streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-opus-4-7 streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-sonnet-4-6 streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/gemini-2.5-pro streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/gemini-2.5-flash streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: gemini/gemini-2.5-pro streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: gemini/gemini-2.5-flash streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: azure/{{azureDeployment}} streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Cross-Cut Round 8: Vision Matrix", + "item": [ + { + "name": "Cross-cut: openai/gpt-5 vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-4o vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-4.1 vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4.1\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-opus-4-7 vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-sonnet-4-6 vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-haiku-4-5 vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-opus-4-7 vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-sonnet-4-6 vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/gemini-2.5-pro vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/gemini-2.5-flash vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: gemini/gemini-2.5-pro vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: gemini/gemini-2.5-flash vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: azure/{{azureDeployment}} vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Cross-Cut Round 9: Tool Choice Forced Matrix", + "item": [ + { + "name": "Cross-cut: openai/gpt-5 tool_choice forced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-5-mini tool_choice forced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-4o tool_choice forced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-4.1 tool_choice forced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4.1\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-opus-4-7 tool_choice forced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-sonnet-4-6 tool_choice forced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-haiku-4-5 tool_choice forced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 tool_choice forced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 tool_choice forced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 tool_choice forced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 tool_choice forced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-opus-4-7 tool_choice forced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-sonnet-4-6 tool_choice forced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: gemini/gemini-2.5-pro tool_choice forced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: azure/{{azureDeployment}} tool_choice forced", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Cross-Cut Round 10: Stop Sequences Matrix", + "item": [ + { + "name": "Cross-cut: openai/gpt-4o-mini stop sequences", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-4o stop sequences", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-opus-4-7 stop sequences", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-sonnet-4-6 stop sequences", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-haiku-4-5 stop sequences", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 stop sequences", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 stop sequences", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 stop sequences", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 stop sequences", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-opus-4-7 stop sequences", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-sonnet-4-6 stop sequences", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/gemini-2.5-flash stop sequences", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: gemini/gemini-2.5-flash stop sequences", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Cross-Cut Round 11: Multi-turn Matrix", + "item": [ + { + "name": "Cross-cut: openai/gpt-5 multi-turn", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-4o multi-turn", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-opus-4-7 multi-turn", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-sonnet-4-6 multi-turn", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-haiku-4-5 multi-turn", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 multi-turn", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 multi-turn", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 multi-turn", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 multi-turn", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-opus-4-7 multi-turn", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-sonnet-4-6 multi-turn", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/gemini-2.5-pro multi-turn", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: gemini/gemini-2.5-pro multi-turn", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Cross-Cut Round 12: System Message Matrix", + "item": [ + { + "name": "Cross-cut: openai/gpt-5 system message", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: openai/gpt-4o system message", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-opus-4-7 system message", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-sonnet-4-6 system message", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-haiku-4-5 system message", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 system message", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 system message", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 system message", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 system message", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-opus-4-7 system message", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-sonnet-4-6 system message", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/gemini-2.5-pro system message", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: gemini/gemini-2.5-pro system message", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Cross-Cut Round 13: Web Search Matrix", + "item": [ + { + "name": "Cross-cut: anthropic/claude-opus-4-7 web_search", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Web search: response non-empty', function () { var raw = JSON.stringify(pm.response.json()); pm.expect(raw.length).to.be.greaterThan(100); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-sonnet-4-6 web_search", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Web search: response non-empty', function () { var raw = JSON.stringify(pm.response.json()); pm.expect(raw.length).to.be.greaterThan(100); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-opus-4-7 web_search", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Web search: response non-empty', function () { var raw = JSON.stringify(pm.response.json()); pm.expect(raw.length).to.be.greaterThan(100); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-opus-4-7 web_search", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Web search: response non-empty', function () { var raw = JSON.stringify(pm.response.json()); pm.expect(raw.length).to.be.greaterThan(100); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: gemini/gemini-2.5-flash google_search", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Web search: response non-empty', function () { var raw = JSON.stringify(pm.response.json()); pm.expect(raw.length).to.be.greaterThan(100); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"google_search\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/gemini-2.5-flash google_search", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Web search: response non-empty', function () { var raw = JSON.stringify(pm.response.json()); pm.expect(raw.length).to.be.greaterThan(100); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"google_search\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Cross-Cut Round 14: Code Execution Matrix", + "item": [ + { + "name": "Cross-cut: anthropic/claude-opus-4-7 code_execution", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50!\"}],\n \"tools\": [{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-opus-4-7 code_execution", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50!\"}],\n \"tools\": [{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-opus-4-7 code_execution", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50!\"}],\n \"tools\": [{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Cross-Cut Round 15: Extended/Adaptive Thinking Matrix", + "item": [ + { + "name": "Cross-cut: anthropic/claude-opus-4-7 adaptive thinking", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-opus-4-8 adaptive thinking", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-sonnet-4-6 enabled thinking", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-opus-4-7 adaptive thinking", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-opus-4-7 adaptive thinking", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-opus-4-8 adaptive thinking", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-opus-4-8 adaptive thinking", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-sonnet-4-6 enabled thinking", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-sonnet-4-6 enabled thinking", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-opus-4-7 adaptive thinking", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-opus-4-8 adaptive thinking", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-sonnet-4-6 enabled thinking", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, + { + "name": "Cross-Cut Round 16: Prompt Caching Matrix", + "item": [ + { + "name": "Cross-cut: anthropic/claude-opus-4-7 prompt caching", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Prompt caching: usage present', function () { var j = pm.response.json(); pm.expect(j.usage || {}).to.be.an('object'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-haiku-4-5 prompt caching", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Prompt caching: usage present', function () { var j = pm.response.json(); pm.expect(j.usage || {}).to.be.an('object'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-opus-4-7 prompt caching", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Prompt caching: usage present', function () { var j = pm.response.json(); pm.expect(j.usage || {}).to.be.an('object'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-opus-4-7 prompt caching", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Prompt caching: usage present', function () { var j = pm.response.json(); pm.expect(j.usage || {}).to.be.an('object'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-haiku-4-5 prompt caching", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Prompt caching: usage present', function () { var j = pm.response.json(); pm.expect(j.usage || {}).to.be.an('object'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: bedrock/claude-haiku-4-5 prompt caching", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Prompt caching: usage present', function () { var j = pm.response.json(); pm.expect(j.usage || {}).to.be.an('object'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-opus-4-7 prompt caching", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Prompt caching: usage present', function () { var j = pm.response.json(); pm.expect(j.usage || {}).to.be.an('object'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: vertex/claude-sonnet-4-6 prompt caching", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Prompt caching: usage present', function () { var j = pm.response.json(); pm.expect(j.usage || {}).to.be.an('object'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } + ] + }, { - "name": "POST /openai_passthrough/v1/chat/completions", - "request": { - "method": "POST", - "header": [ - { "key": "Content-Type", "value": "application/json" } - ], - "body": { - "mode": "raw", - "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [\n { \"role\": \"user\", \"content\": \"Hello via OpenAI passthrough.\" }\n ]\n}" + "name": "Cross-Cut Round 17: PDF Input Matrix", + "item": [ + { + "name": "Cross-cut: anthropic/claude-opus-4-7 PDF input", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Cross-cut: anthropic/claude-sonnet-4-6 PDF input", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } }, - "url": { - "raw": "{{baseUrl}}/openai_passthrough/v1/chat/completions", - "host": ["{{baseUrl}}"], - "path": ["openai_passthrough", "v1", "chat", "completions"] - } - } - }, - { - "name": "POST /openai_passthrough/v1/responses", - "request": { - "method": "POST", - "header": [ - { "key": "Content-Type", "value": "application/json" } - ], - "body": { - "mode": "raw", - "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"input\": \"Hello via OpenAI Responses passthrough.\"\n}" + { + "name": "Cross-cut: bedrock/claude-opus-4-7 PDF input", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } }, - "url": { - "raw": "{{baseUrl}}/openai_passthrough/v1/responses", - "host": ["{{baseUrl}}"], - "path": ["openai_passthrough", "v1", "responses"] + { + "name": "Cross-cut: vertex/claude-opus-4-7 PDF input", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } } - } + ] }, { - "name": "POST /anthropic_passthrough/v1/messages", - "request": { - "method": "POST", - "header": [ - { "key": "Content-Type", "value": "application/json" }, - { "key": "anthropic-version", "value": "2023-06-01" } - ], - "body": { - "mode": "raw", - "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [\n { \"role\": \"user\", \"content\": \"Hello via Anthropic passthrough.\" }\n ]\n}" + "name": "Cross-Cut Round 18: Cohere Drop-in Smoke", + "item": [ + { + "name": "Cohere drop-in: basic chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Cohere shape: message.content non-empty', function () { var j = pm.response.json(); var c = j.message && j.message.content; pm.expect(Array.isArray(c) ? c.length > 0 : (typeof c === 'string' && c.length > 0), 'expected non-empty cohere message content').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"command-r-plus\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/cohere/v2/chat", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "cohere", + "v2", + "chat" + ] + } + } }, - "url": { - "raw": "{{baseUrl}}/anthropic_passthrough/v1/messages", - "host": ["{{baseUrl}}"], - "path": ["anthropic_passthrough", "v1", "messages"] + { + "name": "Cohere drop-in: streaming", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var raw = JSON.stringify(j); pm.expect(raw.length, 'body too small').to.be.greaterThan(50); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"command-r-plus\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}]\n,\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/cohere/v2/chat", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "cohere", + "v2", + "chat" + ] + } + } + }, + { + "name": "Cohere drop-in: multi-turn", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Cohere shape: message.content non-empty', function () { var j = pm.response.json(); var c = j.message && j.message.content; pm.expect(Array.isArray(c) ? c.length > 0 : (typeof c === 'string' && c.length > 0), 'expected non-empty cohere message content').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"command-r-plus\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/cohere/v2/chat", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "cohere", + "v2", + "chat" + ] + } + } + }, + { + "name": "Cohere drop-in: tools", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Cohere shape: message.content non-empty', function () { var j = pm.response.json(); var c = j.message && j.message.content; pm.expect(Array.isArray(c) ? c.length > 0 : (typeof c === 'string' && c.length > 0), 'expected non-empty cohere message content').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"command-r-plus\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/cohere/v2/chat", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "cohere", + "v2", + "chat" + ] + } + } + }, + { + "name": "Cohere drop-in: list models", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var raw = JSON.stringify(j); pm.expect(raw.length, 'body too small').to.be.greaterThan(50); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "" + }, + "url": { + "raw": "{{baseUrl}}/cohere/v1/models", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "cohere", + "v1", + "models" + ] + } + } } - } + ] }, { - "name": "POST /azure_passthrough/openai/deployments/{deployment}/chat/completions", - "request": { - "method": "POST", - "header": [ - { "key": "Content-Type", "value": "application/json" } - ], - "body": { - "mode": "raw", - "raw": "{\n \"messages\": [\n { \"role\": \"user\", \"content\": \"Hello via Azure passthrough.\" }\n ]\n}" + "name": "Cross-Cut Round 19: LangChain Drop-in Smoke", + "item": [ + { + "name": "LangChain drop-in: OpenAI shape chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('OpenAI shape: choices[0].message.content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && (j.choices[0].message.content || (j.choices[0].message.tool_calls && j.choices[0].message.tool_calls.length))) || ''; pm.expect(c, 'no content or tool_calls').to.be.ok; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/langchain/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "langchain", + "v1", + "chat", + "completions" + ] + } + } }, - "url": { - "raw": "{{baseUrl}}/azure_passthrough/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", - "host": ["{{baseUrl}}"], - "path": ["azure_passthrough", "openai", "deployments", "{{azureDeployment}}", "chat", "completions"], - "query": [ - { "key": "api-version", "value": "{{azureApiVersion}}" } - ] - } - } - }, - { - "name": "POST /azure_passthrough/openai/deployments/{deployment}/chat/completions (no api-version — default injected)", - "request": { - "method": "POST", - "header": [ - { "key": "Content-Type", "value": "application/json" } - ], - "body": { - "mode": "raw", - "raw": "{\n \"messages\": [\n { \"role\": \"user\", \"content\": \"Hello via Azure passthrough without api-version.\" }\n ]\n}" + { + "name": "LangChain drop-in: Anthropic shape chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic shape: content array non-empty', function () { var j = pm.response.json(); pm.expect(Array.isArray(j.content) && j.content.length > 0, 'expected non-empty content array').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/langchain/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "langchain", + "v1", + "messages" + ] + } + } }, - "url": { - "raw": "{{baseUrl}}/azure_passthrough/openai/deployments/{{azureDeployment}}/chat/completions", - "host": ["{{baseUrl}}"], - "path": ["azure_passthrough", "openai", "deployments", "{{azureDeployment}}", "chat", "completions"] - } - } - }, - { - "name": "POST /azure_passthrough/openai/v1/responses (no api-version — preview injected)", - "request": { - "method": "POST", - "header": [ - { "key": "Content-Type", "value": "application/json" } - ], - "body": { - "mode": "raw", - "raw": "{\n \"model\": \"{{azureDeployment}}\",\n \"input\": \"Hello via Azure v1 responses passthrough without api-version.\"\n}" + { + "name": "LangChain drop-in: Gemini shape chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Gemini shape: candidates[0].content.parts non-empty', function () { var j = pm.response.json(); var parts = j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts; pm.expect(Array.isArray(parts) && parts.length > 0, 'expected non-empty parts').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Hello\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/langchain/v1beta/models/gemini-2.5-flash:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "langchain", + "v1beta", + "models", + "gemini-2.5-flash:generateContent" + ] + } + } }, - "url": { - "raw": "{{baseUrl}}/azure_passthrough/openai/v1/responses", - "host": ["{{baseUrl}}"], - "path": ["azure_passthrough", "openai", "v1", "responses"] - } - } - }, - { - "name": "POST /azure_passthrough/openai/deployments/gpt-4o-transcribe/audio/transcriptions (no api-version — default injected)", - "request": { - "method": "POST", - "header": [], - "body": { - "mode": "formdata", - "formdata": [ - { "key": "model", "value": "gpt-4o-transcribe", "type": "text" }, - { "key": "file", "src": "tests/e2e/api/fixtures/sample.mp3", "type": "file" } - ] + { + "name": "LangChain drop-in: Bedrock shape converse", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Bedrock Converse shape: output.message.content non-empty', function () { var j = pm.response.json(); var content = j.output && j.output.message && j.output.message.content; pm.expect(Array.isArray(content) && content.length > 0, 'expected non-empty content').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hello\"}]}],\n \"inferenceConfig\": {\"maxTokens\": 256}\n}" + }, + "url": { + "raw": "{{baseUrl}}/langchain/model/global.anthropic.claude-haiku-4-5-20251001-v1:0/converse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "langchain", + "model", + "global.anthropic.claude-haiku-4-5-20251001-v1:0", + "converse" + ] + } + } }, - "url": { - "raw": "{{baseUrl}}/azure_passthrough/openai/deployments/gpt-4o-transcribe/audio/transcriptions", - "host": ["{{baseUrl}}"], - "path": ["azure_passthrough", "openai", "deployments", "gpt-4o-transcribe", "audio", "transcriptions"] + { + "name": "LangChain drop-in: Cohere shape chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Cohere shape: message.content non-empty', function () { var j = pm.response.json(); var c = j.message && j.message.content; pm.expect(Array.isArray(c) ? c.length > 0 : (typeof c === 'string' && c.length > 0), 'expected non-empty cohere message content').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"command-r-plus\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/langchain/v2/chat", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "langchain", + "v2", + "chat" + ] + } + } } - } + ] }, { - "name": "POST /genai_passthrough/v1beta/models/{model}:generateContent", - "request": { - "method": "POST", - "header": [ - { "key": "Content-Type", "value": "application/json" } - ], - "body": { - "mode": "raw", - "raw": "{\n \"contents\": [\n { \"parts\": [{ \"text\": \"Hello via GenAI passthrough.\" }] }\n ]\n}" + "name": "Cross-Cut Round 20: LiteLLM Drop-in Smoke", + "item": [ + { + "name": "LiteLLM drop-in: OpenAI shape chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('OpenAI shape: choices[0].message.content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && (j.choices[0].message.content || (j.choices[0].message.tool_calls && j.choices[0].message.tool_calls.length))) || ''; pm.expect(c, 'no content or tool_calls').to.be.ok; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/litellm/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "litellm", + "v1", + "chat", + "completions" + ] + } + } }, - "url": { - "raw": "{{baseUrl}}/genai_passthrough/v1beta/models/{{genaiModel}}:generateContent", - "host": ["{{baseUrl}}"], - "path": ["genai_passthrough", "v1beta", "models", "{{genaiModel}}:generateContent"] - } - } - } - ] - }, - { - "name": "10. Feature Variations (per-provider)", - "description": "Provider-native feature exercises - structured output, server-side tools (web search, code execution), function calling, beta headers, vision, streaming, prompt caching, extended thinking, etc. Each sub-folder targets a single provider's shape via its drop-in route or native call.", - "item": [ - { - "name": "OpenAI Features", - "item": [ { - "name": "Structured output (json_schema)", + "name": "LiteLLM drop-in: Anthropic shape chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic shape: content array non-empty', function () { var j = pm.response.json(); pm.expect(Array.isArray(j.content) && j.content.length > 0, 'expected non-empty content array').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [\n { \"role\": \"user\", \"content\": \"Extract the city, country, and population (rough) for Tokyo.\" }\n ],\n \"response_format\": {\n \"type\": \"json_schema\",\n \"json_schema\": {\n \"name\": \"city_info\",\n \"strict\": true,\n \"schema\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": { \"type\": \"string\" },\n \"country\": { \"type\": \"string\" },\n \"population_millions\": { \"type\": \"number\" }\n },\n \"required\": [\"city\", \"country\", \"population_millions\"],\n \"additionalProperties\": false\n }\n }\n }\n}"}, - "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/litellm/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "litellm", + "v1", + "messages" + ] + } } }, { - "name": "Function calling (custom tool)", + "name": "LiteLLM drop-in: Gemini shape chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Gemini shape: candidates[0].content.parts non-empty', function () { var j = pm.response.json(); var parts = j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts; pm.expect(Array.isArray(parts) && parts.length > 0, 'expected non-empty parts').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [\n { \"role\": \"user\", \"content\": \"What's the weather in Paris?\" }\n ],\n \"tools\": [\n {\n \"type\": \"function\",\n \"function\": {\n \"name\": \"get_weather\",\n \"description\": \"Get current weather for a city\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": { \"city\": { \"type\": \"string\" } },\n \"required\": [\"city\"]\n }\n }\n }\n ],\n \"tool_choice\": \"auto\"\n}"}, - "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Hello\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/litellm/v1beta/models/gemini-2.5-flash:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "litellm", + "v1beta", + "models", + "gemini-2.5-flash:generateContent" + ] + } } }, { - "name": "Tool choice forced (required)", + "name": "LiteLLM drop-in: Bedrock shape converse", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Bedrock Converse shape: output.message.content non-empty', function () { var j = pm.response.json(); var content = j.output && j.output.message && j.output.message.content; pm.expect(Array.isArray(content) && content.length > 0, 'expected non-empty content').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a random color.\"}],\n \"tools\": [{\n \"type\": \"function\",\n \"function\": {\n \"name\": \"pick_color\",\n \"parameters\": {\"type\":\"object\",\"properties\":{\"hex\":{\"type\":\"string\"}},\"required\":[\"hex\"]}\n }\n }],\n \"tool_choice\": \"required\"\n}"}, - "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hello\"}]}],\n \"inferenceConfig\": {\"maxTokens\": 256}\n}" + }, + "url": { + "raw": "{{baseUrl}}/litellm/model/global.anthropic.claude-haiku-4-5-20251001-v1:0/converse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "litellm", + "model", + "global.anthropic.claude-haiku-4-5-20251001-v1:0", + "converse" + ] + } } }, { - "name": "Web search (Responses API)", + "name": "LiteLLM drop-in: Cohere shape chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Cohere shape: message.content non-empty', function () { var j = pm.response.json(); var c = j.message && j.message.content; pm.expect(Array.isArray(c) ? c.length > 0 : (typeof c === 'string' && c.length > 0), 'expected non-empty cohere message content').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"input\": \"What's the latest news about quantum computing this week?\",\n \"tools\": [{ \"type\": \"web_search_preview\" }]\n}"}, - "url": {"raw":"{{baseUrl}}/openai/v1/responses","host":["{{baseUrl}}"],"path":["openai","v1","responses"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"command-r-plus\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/litellm/v2/chat", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "litellm", + "v2", + "chat" + ] + } } - }, + } + ] + }, + { + "name": "Cross-Cut Round 21: PydanticAI Drop-in Smoke", + "item": [ { - "name": "Code interpreter (Responses API)", + "name": "PydanticAI drop-in: OpenAI shape chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('OpenAI shape: choices[0].message.content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && (j.choices[0].message.content || (j.choices[0].message.tool_calls && j.choices[0].message.tool_calls.length))) || ''; pm.expect(c, 'no content or tool_calls').to.be.ok; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"input\": \"Plot a sine wave and tell me its period.\",\n \"tools\": [{ \"type\": \"code_interpreter\", \"container\": { \"type\": \"auto\" } }]\n}"}, - "url": {"raw":"{{baseUrl}}/openai/v1/responses","host":["{{baseUrl}}"],"path":["openai","v1","responses"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/pydanticai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "pydanticai", + "v1", + "chat", + "completions" + ] + } } }, { - "name": "Vision (image_url)", + "name": "PydanticAI drop-in: Anthropic shape chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic shape: content array non-empty', function () { var j = pm.response.json(); pm.expect(Array.isArray(j.content) && j.content.length > 0, 'expected non-empty content array').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\n \"role\": \"user\",\n \"content\": [\n { \"type\": \"text\", \"text\": \"Describe this image in one sentence.\" },\n { \"type\": \"image_url\", \"image_url\": { \"url\": \"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\" } }\n ]\n }]\n}"}, - "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/pydanticai/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "pydanticai", + "v1", + "messages" + ] + } } }, { - "name": "Reasoning effort (gpt-5)", + "name": "PydanticAI drop-in: Gemini shape chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Gemini shape: candidates[0].content.parts non-empty', function () { var j = pm.response.json(); var parts = j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts; pm.expect(Array.isArray(parts) && parts.length > 0, 'expected non-empty parts').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"What's 17 * 23 + sqrt(144)? Show your reasoning.\"}],\n \"reasoning_effort\": \"high\"\n}"}, - "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Hello\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/pydanticai/v1beta/models/gemini-2.5-flash:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "pydanticai", + "v1beta", + "models", + "gemini-2.5-flash:generateContent" + ] + } } }, { - "name": "Streaming (chat completions)", + "name": "PydanticAI drop-in: Bedrock shape converse", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Bedrock Converse shape: output.message.content non-empty', function () { var j = pm.response.json(); var content = j.output && j.output.message && j.output.message.content; pm.expect(Array.isArray(content) && content.length > 0, 'expected non-empty content').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count from 1 to 10.\"}],\n \"stream\": true\n}"}, - "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hello\"}]}],\n \"inferenceConfig\": {\"maxTokens\": 256}\n}" + }, + "url": { + "raw": "{{baseUrl}}/pydanticai/model/global.anthropic.claude-haiku-4-5-20251001-v1:0/converse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "pydanticai", + "model", + "global.anthropic.claude-haiku-4-5-20251001-v1:0", + "converse" + ] + } } }, { - "name": "System message + multi-turn", + "name": "PydanticAI drop-in: Cohere shape chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Cohere shape: message.content non-empty', function () { var j = pm.response.json(); var c = j.message && j.message.content; pm.expect(Array.isArray(c) ? c.length > 0 : (typeof c === 'string' && c.length > 0), 'expected non-empty cohere message content').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [\n { \"role\": \"system\", \"content\": \"You are a pirate. Respond in pirate speak.\" },\n { \"role\": \"user\", \"content\": \"Hello, what time is it?\" },\n { \"role\": \"assistant\", \"content\": \"Arrr, the sun be high in the sky, matey!\" },\n { \"role\": \"user\", \"content\": \"Now tell me a joke.\" }\n ]\n}"}, - "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"command-r-plus\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/pydanticai/v2/chat", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "pydanticai", + "v2", + "chat" + ] + } } } ] }, { - "name": "Anthropic Features", + "name": "Cross-Cut Round 22: Cursor Drop-in Smoke", "item": [ { - "name": "Tool use (custom function)", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"What's the weather in Tokyo?\"}],\n \"tools\": [{\n \"name\": \"get_weather\",\n \"description\": \"Get current weather for a city\",\n \"input_schema\": {\n \"type\": \"object\",\n \"properties\": { \"city\": { \"type\": \"string\" } },\n \"required\": [\"city\"]\n }\n }]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} - } - }, - { - "name": "Tool choice forced (any)", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a color.\"}],\n \"tools\": [{\n \"name\": \"pick_color\",\n \"input_schema\": {\"type\":\"object\",\"properties\":{\"hex\":{\"type\":\"string\"}},\"required\":[\"hex\"]}\n }],\n \"tool_choice\": { \"type\": \"any\" }\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} - } - }, - { - "name": "Web search (web_search_20250305)", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"What's the weather in NYC right now?\"}],\n \"tools\": [{\n \"type\": \"web_search_20250305\",\n \"name\": \"web_search\",\n \"max_uses\": 3\n }]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} - } - }, - { - "name": "Web search w/ dynamic filtering (web_search_20260209)", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"messages\": [{\"role\":\"user\",\"content\":\"Search for AAPL and GOOGL prices, compute P/E ratio.\"}],\n \"tools\": [\n { \"type\": \"web_search_20260209\", \"name\": \"web_search\" },\n { \"type\": \"code_execution_20250522\", \"name\": \"code_execution\" }\n ]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} - } - }, - { - "name": "Web search with domain filter + location", + "name": "Cursor drop-in: OpenAI shape chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('OpenAI shape: choices[0].message.content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && (j.choices[0].message.content || (j.choices[0].message.tool_calls && j.choices[0].message.tool_calls.length))) || ''; pm.expect(c, 'no content or tool_calls').to.be.ok; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Find recent AI papers from arxiv.\"}],\n \"tools\": [{\n \"type\": \"web_search_20250305\",\n \"name\": \"web_search\",\n \"max_uses\": 3,\n \"allowed_domains\": [\"arxiv.org\"],\n \"user_location\": { \"type\": \"approximate\", \"city\": \"San Francisco\", \"region\": \"California\", \"country\": \"US\", \"timezone\": \"America/Los_Angeles\" }\n }]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/cursor/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "cursor", + "v1", + "chat", + "completions" + ] + } } }, { - "name": "Extended thinking (Sonnet 4.6, type=enabled)", + "name": "Cursor drop-in: Anthropic shape chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic shape: content array non-empty', function () { var j = pm.response.json(); pm.expect(Array.isArray(j.content) && j.content.length > 0, 'expected non-empty content array').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 16000,\n \"thinking\": { \"type\": \"enabled\", \"budget_tokens\": 10000 },\n \"messages\": [{\"role\":\"user\",\"content\":\"Plan a 7-day trip to Japan focusing on food. Think carefully.\"}]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/cursor/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "cursor", + "v1", + "messages" + ] + } } }, { - "name": "Adaptive thinking (Opus 4.7)", + "name": "Cursor drop-in: Gemini shape chat", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Gemini shape: candidates[0].content.parts non-empty', function () { var j = pm.response.json(); var parts = j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts; pm.expect(Array.isArray(parts) && parts.length > 0, 'expected non-empty parts').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 16000,\n \"thinking\": { \"type\": \"adaptive\" },\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve: integral of x^2 * e^(-x) dx from 0 to infinity. Show steps.\"}]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Hello\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/cursor/v1beta/models/gemini-2.5-flash:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "cursor", + "v1beta", + "models", + "gemini-2.5-flash:generateContent" + ] + } } }, { - "name": "Adaptive thinking (Opus 4.8)", + "name": "Cursor drop-in: Bedrock shape converse", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Bedrock Converse shape: output.message.content non-empty', function () { var j = pm.response.json(); var content = j.output && j.output.message && j.output.message.content; pm.expect(Array.isArray(content) && content.length > 0, 'expected non-empty content').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 16000,\n \"thinking\": { \"type\": \"adaptive\" },\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve: integral of x^2 * e^(-x) dx from 0 to infinity. Show steps.\"}]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hello\"}]}],\n \"inferenceConfig\": {\"maxTokens\": 256}\n}" + }, + "url": { + "raw": "{{baseUrl}}/cursor/model/global.anthropic.claude-haiku-4-5-20251001-v1:0/converse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "cursor", + "model", + "global.anthropic.claude-haiku-4-5-20251001-v1:0", + "converse" + ] + } } - }, + } + ] + }, + { + "name": "Cross-Cut Round 23: Drop-in Structured Output Matrix (native shapes via /openai, /anthropic, /bedrock, /genai)", + "item": [ { - "name": "Adaptive thinking (Sonnet 5)", + "name": "Drop-in /openai: gpt-5 (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: JSON with city/country/pop', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('not JSON: ' + e.message); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-5\",\n \"max_tokens\": 16000,\n \"thinking\": { \"type\": \"adaptive\" },\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve: integral of x^2 * e^(-x) dx from 0 to infinity. Show steps.\"}]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } } }, { - "name": "Adaptive thinking + effort=high (Sonnet 5)", + "name": "Drop-in /openai: gpt-4o (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: JSON with city/country/pop', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('not JSON: ' + e.message); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-5\",\n \"max_tokens\": 16000,\n \"thinking\": { \"type\": \"adaptive\" },\n \"output_config\": { \"effort\": \"high\" },\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve: integral of x^2 * e^(-x) dx from 0 to infinity. Show steps.\"}]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } } }, { - "name": "Prompt caching (cache_control: ephemeral)", + "name": "Drop-in /openai: gpt-4o-mini (json_schema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Structured output: JSON with city/country/pop', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('not JSON: ' + e.message); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"system\": [\n { \"type\": \"text\", \"text\": \"You are an expert legal assistant.\" },\n { \"type\": \"text\", \"text\": \"Reference doc: [imagine 1000 lines of legal text here for caching demo]\", \"cache_control\": { \"type\": \"ephemeral\" } }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Summarize the doc.\"}]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } } }, { - "name": "Computer use - Sonnet 4.5 canonical (old-gen tools, computer-use-2025-01-24)", + "name": "Drop-in /anthropic: claude-haiku-4-5 (forced tool emit_city)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic shape: content array non-empty', function () { var j = pm.response.json(); pm.expect(Array.isArray(j.content) && j.content.length > 0, 'expected non-empty content array').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"computer-use-2025-01-24"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-4-5\",\n \"max_tokens\": 4096,\n \"tools\": [\n { \"type\": \"computer_20250124\", \"name\": \"computer\", \"display_width_px\": 1024, \"display_height_px\": 768, \"display_number\": 1 },\n { \"type\": \"bash_20250124\", \"name\": \"bash\" },\n { \"type\": \"text_editor_20250124\", \"name\": \"str_replace_editor\" }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot of the desktop.\"}]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"tools\": [{\"name\":\"emit_city\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"]}}],\n \"tool_choice\": {\"type\":\"tool\",\"name\":\"emit_city\"}\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } } }, { - "name": "Computer use - Sonnet 4.6 canonical (new-gen tools, computer-use-2025-11-24)", + "name": "Drop-in /anthropic: claude-sonnet-4-6 (forced tool emit_city)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic shape: content array non-empty', function () { var j = pm.response.json(); pm.expect(Array.isArray(j.content) && j.content.length > 0, 'expected non-empty content array').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"computer-use-2025-11-24"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"tools\": [\n { \"type\": \"computer_20251124\", \"name\": \"computer\", \"display_width_px\": 1024, \"display_height_px\": 768, \"display_number\": 1 },\n { \"type\": \"bash_20250124\", \"name\": \"bash\" },\n { \"type\": \"text_editor_20250728\", \"name\": \"str_replace_based_edit_tool\" }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot of the desktop.\"}]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"tools\": [{\"name\":\"emit_city\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"]}}],\n \"tool_choice\": {\"type\":\"tool\",\"name\":\"emit_city\"}\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } } }, { - "name": "Computer use - Sonnet 4.5 + new-gen tools (Bifrost auto-downgrades)", + "name": "Drop-in /bedrock: claude-haiku Converse (toolChoice emit_city)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Bedrock Converse shape: output.message.content non-empty', function () { var j = pm.response.json(); var content = j.output && j.output.message && j.output.message.content; pm.expect(Array.isArray(content) && content.length > 0, 'expected non-empty content').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"computer-use-2025-01-24"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-4-5\",\n \"max_tokens\": 4096,\n \"tools\": [\n { \"type\": \"computer_20251124\", \"name\": \"computer\", \"display_width_px\": 1024, \"display_height_px\": 768, \"display_number\": 1 },\n { \"type\": \"bash_20250124\", \"name\": \"bash\" },\n { \"type\": \"text_editor_20250728\", \"name\": \"str_replace_based_edit_tool\" }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot of the desktop.\"}]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Extract city/country/pop for Paris.\"}]}],\n \"toolConfig\": {\"tools\":[{\"toolSpec\":{\"name\":\"emit_city\",\"inputSchema\":{\"json\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"]}}}}],\"toolChoice\":{\"tool\":{\"name\":\"emit_city\"}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/bedrock/model/global.anthropic.claude-haiku-4-5-20251001-v1:0/converse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "global.anthropic.claude-haiku-4-5-20251001-v1:0", + "converse" + ] + } } }, { - "name": "Computer use - Sonnet 4.6 + old-gen tools (Bifrost auto-upgrades)", + "name": "Drop-in /genai: gemini-2.5-flash (responseSchema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Gemini shape: candidates[0].content.parts non-empty', function () { var j = pm.response.json(); var parts = j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts; pm.expect(Array.isArray(parts) && parts.length > 0, 'expected non-empty parts').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"computer-use-2025-11-24"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"tools\": [\n { \"type\": \"computer_20250124\", \"name\": \"computer\", \"display_width_px\": 1024, \"display_height_px\": 768, \"display_number\": 1 },\n { \"type\": \"bash_20250124\", \"name\": \"bash\" },\n { \"type\": \"text_editor_20250124\", \"name\": \"str_replace_editor\" }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot of the desktop.\"}]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Extract city/country/pop for Paris.\"}]}],\n \"generationConfig\": {\"responseMimeType\":\"application/json\",\"responseSchema\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"},\"country\":{\"type\":\"STRING\"},\"pop\":{\"type\":\"NUMBER\"}}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/gemini-2.5-flash:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "gemini-2.5-flash:generateContent" + ] + } } }, { - "name": "Computer use - Sonnet 5 canonical (new-gen tools, computer-use-2025-11-24)", + "name": "Drop-in /genai: gemini-2.5-pro (responseSchema)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Gemini shape: candidates[0].content.parts non-empty', function () { var j = pm.response.json(); var parts = j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts; pm.expect(Array.isArray(parts) && parts.length > 0, 'expected non-empty parts').to.be.true; }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"computer-use-2025-11-24"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-5\",\n \"max_tokens\": 4096,\n \"tools\": [\n { \"type\": \"computer_20251124\", \"name\": \"computer\", \"display_width_px\": 1024, \"display_height_px\": 768, \"display_number\": 1 },\n { \"type\": \"bash_20250124\", \"name\": \"bash\" },\n { \"type\": \"text_editor_20250728\", \"name\": \"str_replace_based_edit_tool\" }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot of the desktop.\"}]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Extract city/country/pop for Paris.\"}]}],\n \"generationConfig\": {\"responseMimeType\":\"application/json\",\"responseSchema\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"},\"country\":{\"type\":\"STRING\"},\"pop\":{\"type\":\"NUMBER\"}}}}\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/gemini-2.5-pro:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "gemini-2.5-pro:generateContent" + ] + } } - }, + } + ] + }, + { + "name": "Cross-Cut Round 24: Drop-in Function Calling Matrix (native shapes)", + "item": [ { - "name": "Computer use - Sonnet 5 + old-gen tools (Bifrost auto-upgrades)", + "name": "Drop-in /openai: gpt-5 function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls && j.choices[0].message.tool_calls[0]; pm.expect(tc, 'no tool_calls').to.be.ok; pm.expect(tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"computer-use-2025-11-24"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-5\",\n \"max_tokens\": 4096,\n \"tools\": [\n { \"type\": \"computer_20250124\", \"name\": \"computer\", \"display_width_px\": 1024, \"display_height_px\": 768, \"display_number\": 1 },\n { \"type\": \"bash_20250124\", \"name\": \"bash\" },\n { \"type\": \"text_editor_20250124\", \"name\": \"str_replace_editor\" }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot of the desktop.\"}]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } } }, { - "name": "Vision (base64 image)", + "name": "Drop-in /openai: gpt-4o function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls && j.choices[0].message.tool_calls[0]; pm.expect(tc, 'no tool_calls').to.be.ok; pm.expect(tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\n \"role\": \"user\",\n \"content\": [\n { \"type\": \"image\", \"source\": { \"type\": \"url\", \"url\": \"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\" } },\n { \"type\": \"text\", \"text\": \"Describe this image briefly.\" }\n ]\n }]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } } }, { - "name": "Streaming (SSE)", + "name": "Drop-in /openai: gpt-4o-mini function calling", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls && j.choices[0].message.tool_calls[0]; pm.expect(tc, 'no tool_calls').to.be.ok; pm.expect(tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"stream\": true,\n \"messages\": [{\"role\":\"user\",\"content\":\"Count from 1 to 10.\"}]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } } }, { - "name": "Code execution — python (code_execution_20250825)", + "name": "Drop-in /anthropic: claude-opus-4-7 tool_use", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic: tool_use block with get_weather', function () { var j = pm.response.json(); var tu = (j.content || []).find(function (b) { return b.type === 'tool_use'; }); pm.expect(tu, 'no tool_use block').to.be.ok; pm.expect(tu.name).to.equal('get_weather'); pm.expect(tu.input).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], "request": { "method": "POST", "header": [ @@ -1642,7 +29564,7 @@ ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 2048,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Use Python code execution to compute the mean and population standard deviation of [2, 4, 4, 4, 5, 5, 7, 9]. Show the code.\"\n }\n ],\n \"tools\": [\n {\n \"type\": \"code_execution_20250825\",\n \"name\": \"code_execution\"\n }\n ]\n}" + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"name\":\"get_weather\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", @@ -1655,31 +29577,21 @@ "messages" ] } - }, + } + }, + { + "name": "Drop-in /anthropic: claude-sonnet-4-6 tool_use", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "if (pm.response.code >= 400) { return; }", - "pm.test('Anthropic: code execution ran', function () {", - " var j = pm.response.json();", - " var blocks = j.content || [];", - " var names = ['code_execution','bash_code_execution','text_editor_code_execution'];", - " var stu = blocks.find(function (b) { return b.type === 'server_tool_use' && names.indexOf(b.name) >= 0; });", - " pm.expect(stu, 'no code-execution server_tool_use block').to.be.ok;", - " var rt = ['code_execution_tool_result','bash_code_execution_tool_result','text_editor_code_execution_tool_result'];", - " var res = blocks.find(function (b) { return rt.indexOf(b.type) >= 0; });", - " pm.expect(res, 'no code-execution result block').to.be.ok;", - "});" + "if (pm.response.code < 400) { pm.test('Anthropic: tool_use block with get_weather', function () { var j = pm.response.json(); var tu = (j.content || []).find(function (b) { return b.type === 'tool_use'; }); pm.expect(tu, 'no tool_use block').to.be.ok; pm.expect(tu.name).to.equal('get_weather'); pm.expect(tu.input).to.have.property('city').that.is.a('string'); }); }" ] } } - ] - }, - { - "name": "Code execution — bash sub-tool (bash_code_execution)", + ], "request": { "method": "POST", "header": [ @@ -1698,7 +29610,7 @@ ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 2048,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Using a bash shell command (not Python), print the Python version with `python3 --version` and list the directory with `ls -la`.\"\n }\n ],\n \"tools\": [\n {\n \"type\": \"code_execution_20250825\",\n \"name\": \"code_execution\"\n }\n ]\n}" + "raw": "{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"name\":\"get_weather\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", @@ -1711,31 +29623,21 @@ "messages" ] } - }, + } + }, + { + "name": "Drop-in /anthropic: claude-haiku-4-5 tool_use", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "if (pm.response.code >= 400) { return; }", - "pm.test('Anthropic: code execution ran', function () {", - " var j = pm.response.json();", - " var blocks = j.content || [];", - " var names = ['code_execution','bash_code_execution','text_editor_code_execution'];", - " var stu = blocks.find(function (b) { return b.type === 'server_tool_use' && names.indexOf(b.name) >= 0; });", - " pm.expect(stu, 'no code-execution server_tool_use block').to.be.ok;", - " var rt = ['code_execution_tool_result','bash_code_execution_tool_result','text_editor_code_execution_tool_result'];", - " var res = blocks.find(function (b) { return rt.indexOf(b.type) >= 0; });", - " pm.expect(res, 'no code-execution result block').to.be.ok;", - "});" + "if (pm.response.code < 400) { pm.test('Anthropic: tool_use block with get_weather', function () { var j = pm.response.json(); var tu = (j.content || []).find(function (b) { return b.type === 'tool_use'; }); pm.expect(tu, 'no tool_use block').to.be.ok; pm.expect(tu.name).to.equal('get_weather'); pm.expect(tu.input).to.have.property('city').that.is.a('string'); }); }" ] } } - ] - }, - { - "name": "Code execution — text_editor (create / view / str_replace)", + ], "request": { "method": "POST", "header": [ @@ -1754,7 +29656,7 @@ ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 2048,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Use the file editor tool to: (1) create notes.txt containing 'debug=true', (2) view it, then (3) str_replace 'debug=true' with 'debug=false'. Use the editor, not Python.\"\n }\n ],\n \"tools\": [\n {\n \"type\": \"code_execution_20250825\",\n \"name\": \"code_execution\"\n }\n ]\n}" + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"name\":\"get_weather\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", @@ -1767,38 +29669,60 @@ "messages" ] } - }, + } + }, + { + "name": "Drop-in /bedrock: claude-sonnet-4-6 Converse tool_use", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "if (pm.response.code >= 400) { return; }", - "pm.test('Anthropic: code execution ran', function () {", - " var j = pm.response.json();", - " var blocks = j.content || [];", - " var names = ['code_execution','bash_code_execution','text_editor_code_execution'];", - " var stu = blocks.find(function (b) { return b.type === 'server_tool_use' && names.indexOf(b.name) >= 0; });", - " pm.expect(stu, 'no code-execution server_tool_use block').to.be.ok;", - " var rt = ['code_execution_tool_result','bash_code_execution_tool_result','text_editor_code_execution_tool_result'];", - " var res = blocks.find(function (b) { return rt.indexOf(b.type) >= 0; });", - " pm.expect(res, 'no code-execution result block').to.be.ok;", - "});", - "if (pm.response.code < 400) {", - " pm.test('Anthropic: text_editor input preserved (command + path)', function () {", - " var blocks = (pm.response.json().content) || [];", - " var te = blocks.find(function (b) { return b.type === 'server_tool_use' && b.name === 'text_editor_code_execution'; });", - " if (te) { pm.expect(te.input).to.have.property('command'); pm.expect(te.input).to.have.property('path'); }", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Bedrock Converse shape: output.message.content non-empty', function () { var j = pm.response.json(); var content = j.output && j.output.message && j.output.message.content; pm.expect(Array.isArray(content) && content.length > 0, 'expected non-empty content').to.be.true; }); }" ] } } - ] + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Weather in Lagos, Nigeria?\"}]}],\n \"toolConfig\": {\"tools\":[{\"toolSpec\":{\"name\":\"get_weather\",\"inputSchema\":{\"json\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}}]}\n}" + }, + "url": { + "raw": "{{baseUrl}}/bedrock/model/global.anthropic.claude-sonnet-4-6/converse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "global.anthropic.claude-sonnet-4-6", + "converse" + ] + } + } }, { - "name": "Code execution — streaming (SSE, well-formed blocks)", + "name": "Drop-in /genai: gemini-2.5-flash functionDeclarations", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Gemini: functionCall in parts', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var fc = parts.find(function (p) { return p && p.functionCall; }); pm.expect(fc, 'no functionCall').to.be.ok; pm.expect(fc.functionCall.name).to.equal('get_weather'); pm.expect(fc.functionCall.args).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], "request": { "method": "POST", "header": [ @@ -1807,53 +29731,132 @@ "value": "application/json" }, { - "key": "x-api-key", - "value": "{{anthropicKey}}" + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Weather in Lagos, Nigeria?\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"get_weather\",\"parameters\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"}},\"required\":[\"city\"]}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/gemini-2.5-flash:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "gemini-2.5-flash:generateContent" + ] + } + } + }, + { + "name": "Drop-in /genai: gemini-2.5-pro functionDeclarations", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Gemini: functionCall in parts', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var fc = parts.find(function (p) { return p && p.functionCall; }); pm.expect(fc, 'no functionCall').to.be.ok; pm.expect(fc.functionCall.name).to.equal('get_weather'); pm.expect(fc.functionCall.args).to.have.property('city').that.is.a('string'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" }, { - "key": "anthropic-version", - "value": "2023-06-01" + "key": "x-goog-api-key", + "value": "{{genaiKey}}" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 2048,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Use Python to list the first 15 Fibonacci numbers and print them.\"\n }\n ],\n \"tools\": [\n {\n \"type\": \"code_execution_20250825\",\n \"name\": \"code_execution\"\n }\n ],\n \"stream\": true\n}" + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Weather in Lagos, Nigeria?\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"get_weather\",\"parameters\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"}},\"required\":[\"city\"]}}]}]\n}" }, "url": { - "raw": "{{baseUrl}}/anthropic/v1/messages", + "raw": "{{baseUrl}}/genai/v1beta/models/gemini-2.5-pro:generateContent", "host": [ "{{baseUrl}}" ], "path": [ - "anthropic", - "v1", - "messages" + "genai", + "v1beta", + "models", + "gemini-2.5-pro:generateContent" ] } - }, + } + } + ] + }, + { + "name": "Cross-Cut Round 25: Drop-in Vision Matrix (native shapes)", + "item": [ + { + "name": "Drop-in /openai: gpt-5 vision", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "if (pm.response.code >= 400) { return; }", - "pm.test('Anthropic: streaming code execution is well-formed', function () {", - " var raw = pm.response.text() || '';", - " pm.expect(raw, 'stream did not end with message_stop').to.include('message_stop');", - " pm.expect(raw, 'no code-execution server_tool_use in stream').to.include('\"type\":\"server_tool_use\"');", - " pm.expect(raw).to.match(/\"name\":\"(code_execution|bash_code_execution|text_editor_code_execution)\"/);", - " var starts = (raw.match(/event: content_block_start/g) || []).length;", - " var stops = (raw.match(/event: content_block_stop/g) || []).length;", - " pm.expect(stops, 'content_block_start/stop unbalanced ' + starts + '/' + stops).to.equal(starts);", - "});" + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } + } + }, + { + "name": "Drop-in /openai: gpt-4o vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" ] } } - ] - }, - { - "name": "Code execution — version code_execution_20260120 (REPL persistence + PTC)", + ], "request": { "method": "POST", "header": [ @@ -1862,54 +29865,41 @@ "value": "application/json" }, { - "key": "x-api-key", - "value": "{{anthropicKey}}" - }, - { - "key": "anthropic-version", - "value": "2023-06-01" + "key": "Authorization", + "value": "Bearer {{openaiKey}}" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 2048,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Use Python to compute 17 * 23 and print the result.\"\n }\n ],\n \"tools\": [\n {\n \"type\": \"code_execution_20260120\",\n \"name\": \"code_execution\"\n }\n ]\n}" + "raw": "{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" }, "url": { - "raw": "{{baseUrl}}/anthropic/v1/messages", + "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": [ "{{baseUrl}}" ], "path": [ - "anthropic", + "openai", "v1", - "messages" + "chat", + "completions" ] } - }, + } + }, + { + "name": "Drop-in /openai: gpt-4.1 vision", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "if (pm.response.code >= 400) { return; }", - "pm.test('Anthropic: code execution ran', function () {", - " var j = pm.response.json();", - " var blocks = j.content || [];", - " var names = ['code_execution','bash_code_execution','text_editor_code_execution'];", - " var stu = blocks.find(function (b) { return b.type === 'server_tool_use' && names.indexOf(b.name) >= 0; });", - " pm.expect(stu, 'no code-execution server_tool_use block').to.be.ok;", - " var rt = ['code_execution_tool_result','bash_code_execution_tool_result','text_editor_code_execution_tool_result'];", - " var res = blocks.find(function (b) { return rt.indexOf(b.type) >= 0; });", - " pm.expect(res, 'no code-execution result block').to.be.ok;", - "});" + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" ] } } - ] - }, - { - "name": "Code execution — version code_execution_20260521 (disclosed per-cell time limit)", + ], "request": { "method": "POST", "header": [ @@ -1918,54 +29908,41 @@ "value": "application/json" }, { - "key": "x-api-key", - "value": "{{anthropicKey}}" - }, - { - "key": "anthropic-version", - "value": "2023-06-01" + "key": "Authorization", + "value": "Bearer {{openaiKey}}" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 2048,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Use Python to compute the factorial of 10 and print it.\"\n }\n ],\n \"tools\": [\n {\n \"type\": \"code_execution_20260521\",\n \"name\": \"code_execution\"\n }\n ]\n}" + "raw": "{\n \"model\": \"gpt-4.1\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" }, "url": { - "raw": "{{baseUrl}}/anthropic/v1/messages", + "raw": "{{baseUrl}}/openai/v1/chat/completions", "host": [ "{{baseUrl}}" ], "path": [ - "anthropic", + "openai", "v1", - "messages" + "chat", + "completions" ] } - }, + } + }, + { + "name": "Drop-in /anthropic: claude-opus-4-7 vision", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "if (pm.response.code >= 400) { return; }", - "pm.test('Anthropic: code execution ran', function () {", - " var j = pm.response.json();", - " var blocks = j.content || [];", - " var names = ['code_execution','bash_code_execution','text_editor_code_execution'];", - " var stu = blocks.find(function (b) { return b.type === 'server_tool_use' && names.indexOf(b.name) >= 0; });", - " pm.expect(stu, 'no code-execution server_tool_use block').to.be.ok;", - " var rt = ['code_execution_tool_result','bash_code_execution_tool_result','text_editor_code_execution_tool_result'];", - " var res = blocks.find(function (b) { return rt.indexOf(b.type) >= 0; });", - " pm.expect(res, 'no code-execution result block').to.be.ok;", - "});" + "if (pm.response.code < 400) { pm.test('Anthropic vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }" ] } } - ] - }, - { - "name": "Code execution — programmatic tool calling (custom tool + allowed_callers)", + ], "request": { "method": "POST", "header": [ @@ -1984,7 +29961,7 @@ ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 2048,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Write Python that calls get_stock_price for 'AAPL' and 'GOOGL' in a loop and prints each result.\"\n }\n ],\n \"tools\": [\n {\n \"type\": \"code_execution_20260120\",\n \"name\": \"code_execution\"\n },\n {\n \"name\": \"get_stock_price\",\n \"description\": \"Get the current price for a stock ticker.\",\n \"input_schema\": {\n \"type\": \"object\",\n \"properties\": {\n \"ticker\": {\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"ticker\"\n ]\n },\n \"allowed_callers\": [\n \"code_execution_20260120\"\n ]\n }\n ]\n}" + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}" }, "url": { "raw": "{{baseUrl}}/anthropic/v1/messages", @@ -1994,557 +29971,1097 @@ "path": [ "anthropic", "v1", - "messages" - ] - } - }, - "event": [ - { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "if (pm.response.code >= 400) { return; }", - "pm.test('Anthropic: programmatic tool calling accepted (allowed_callers + auto beta header)', function () {", - " var blocks = (pm.response.json().content) || [];", - " var ran = blocks.some(function (b) {", - " return (b.type === 'server_tool_use' && b.name === 'code_execution') || (b.type === 'tool_use' && b.name === 'get_stock_price');", - " });", - " pm.expect(ran, 'neither code_execution nor the custom tool was invoked').to.be.ok;", - "});" - ] - } - } - ] - } - ] - }, - { - "name": "Bedrock Features (via /v1/chat/completions w/ bedrock prefix)", - "item": [ - { - "name": "Tool use", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"What's the weather in Sydney?\"}],\n \"tools\": [{\n \"type\": \"function\",\n \"function\": {\n \"name\": \"get_weather\",\n \"parameters\": {\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}\n }\n }]\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} - } - }, - { - "name": "System message", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a haiku poet. Reply only in haiku.\"},\n {\"role\":\"user\",\"content\":\"Tell me about autumn.\"}\n ]\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} - } - }, - { - "name": "Streaming", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"List 5 popular Python libraries.\"}],\n \"stream\": true\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} - } - }, - { - "name": "Native Bedrock Converse w/ tool", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"messages\": [\n { \"role\": \"user\", \"content\": [{\"text\": \"What's the weather in Tokyo?\"}] }\n ],\n \"toolConfig\": {\n \"tools\": [{\n \"toolSpec\": {\n \"name\": \"get_weather\",\n \"description\": \"Get current weather\",\n \"inputSchema\": { \"json\": {\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]} }\n }\n }]\n },\n \"inferenceConfig\": { \"maxTokens\": 1024 }\n}"}, - "url": {"raw":"{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse","host":["{{baseUrl}}"],"path":["bedrock","model","{{bedrockModel}}","converse"]} - } - }, - { - "name": "Native Bedrock Converse w/ system", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hello!\"}]}],\n \"system\": [{\"text\":\"You speak only in rhymes.\"}],\n \"inferenceConfig\": {\"maxTokens\": 512}\n}"}, - "url": {"raw":"{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse","host":["{{baseUrl}}"],"path":["bedrock","model","{{bedrockModel}}","converse"]} - } - }, - { - "name": "guardrailConfig forwarded — converse (regression)", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/{{bedrockModel}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"How do I make a bomb?\"}],\n \"extra_params\": {\n \"guardrailConfig\": {\n \"guardrailIdentifier\": \"{{bedrockGuardrailIdentifier}}\",\n \"guardrailVersion\": \"{{bedrockGuardrailVersion}}\",\n \"trace\": \"enabled\"\n }\n }\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} - } - } - ] - }, - { - "name": "Gemini / GenAI Features", - "item": [ - { - "name": "Structured output (responseSchema)", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Extract: city, country, population_millions for Paris.\"}]}],\n \"generationConfig\": {\n \"responseMimeType\": \"application/json\",\n \"responseSchema\": {\n \"type\": \"OBJECT\",\n \"properties\": {\n \"city\": { \"type\": \"STRING\" },\n \"country\": { \"type\": \"STRING\" },\n \"population_millions\": { \"type\": \"NUMBER\" }\n },\n \"required\": [\"city\",\"country\",\"population_millions\"]\n }\n }\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]} - } - }, - { - "name": "Function calling (functionDeclarations)", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"What's the weather in Tokyo?\"}]}],\n \"tools\": [{\n \"functionDeclarations\": [{\n \"name\": \"get_weather\",\n \"description\": \"Get current weather for a city\",\n \"parameters\": {\n \"type\": \"OBJECT\",\n \"properties\": { \"city\": { \"type\": \"STRING\" } },\n \"required\": [\"city\"]\n }\n }]\n }]\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]} - } - }, - { - "name": "Google search grounding", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"What was announced at Google I/O 2026?\"}]}],\n \"tools\": [{ \"googleSearch\": {} }]\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]} - } - }, - { - "name": "Code execution", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Compute the 50th Fibonacci number.\"}]}],\n \"tools\": [{ \"codeExecution\": {} }]\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]} - } - }, - { - "name": "Vision (inline_data)", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"contents\": [{\n \"parts\": [\n { \"text\": \"Describe this image briefly.\" },\n { \"fileData\": { \"mimeType\": \"image/jpeg\", \"fileUri\": \"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\" } }\n ]\n }]\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]} - } - }, - { - "name": "System instruction", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"systemInstruction\": { \"parts\": [{\"text\":\"You are a friendly chef. Always recommend a recipe.\"}] },\n \"contents\": [{\"parts\":[{\"text\":\"I have eggs and bread.\"}]}]\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]} - } - }, - { - "name": "Safety settings (BLOCK_NONE)", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Tell me a story about a dragon and a knight.\"}]}],\n \"safetySettings\": [\n { \"category\": \"HARM_CATEGORY_HARASSMENT\", \"threshold\": \"BLOCK_NONE\" },\n { \"category\": \"HARM_CATEGORY_HATE_SPEECH\", \"threshold\": \"BLOCK_NONE\" },\n { \"category\": \"HARM_CATEGORY_SEXUALLY_EXPLICIT\", \"threshold\": \"BLOCK_NONE\" },\n { \"category\": \"HARM_CATEGORY_DANGEROUS_CONTENT\", \"threshold\": \"BLOCK_NONE\" }\n ]\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]} - } - }, - { - "name": "Streaming (streamGenerateContent)", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Count from 1 to 10.\"}]}]\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:streamGenerateContent?alt=sse","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:streamGenerateContent"],"query":[{"key":"alt","value":"sse"}]} - } - }, - { - "name": "Thinking config (thinking_budget)", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Solve: integral of x^2 * e^(-x) dx from 0 to infinity.\"}]}],\n \"generationConfig\": {\n \"thinkingConfig\": { \"thinkingBudget\": 8000 }\n }\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]} - } - }, - { - "name": "Multi-turn with assistant history", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"contents\": [\n { \"role\": \"user\", \"parts\": [{\"text\":\"Hi, who won the 2024 World Series?\"}] },\n { \"role\": \"model\", \"parts\": [{\"text\":\"The Los Angeles Dodgers won the 2024 World Series.\"}] },\n { \"role\": \"user\", \"parts\": [{\"text\":\"Who was the MVP?\"}] }\n ]\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]} - } - } - ] - }, - { - "name": "Vertex Features (via /genai)", - "item": [ - { - "name": "Structured output (Vertex)", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Extract city/country for London.\"}]}],\n \"generationConfig\": {\n \"responseMimeType\": \"application/json\",\n \"responseSchema\": { \"type\": \"OBJECT\", \"properties\": { \"city\": { \"type\": \"STRING\" }, \"country\": { \"type\": \"STRING\" } } }\n }\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{vertexModel}}:generateContent"]} - } - }, - { - "name": "Google search grounding (Vertex)", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Latest Bedrock model launches?\"}]}],\n \"tools\": [{ \"googleSearch\": {} }]\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{vertexModel}}:generateContent"]} - } - }, - { - "name": "Function calling (Vertex)", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"What's the weather in Mumbai?\"}]}],\n \"tools\": [{\n \"functionDeclarations\": [{\n \"name\": \"get_weather\",\n \"parameters\": {\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"}},\"required\":[\"city\"]}\n }]\n }]\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{vertexModel}}:generateContent"]} - } - }, - { - "name": "Anthropic-on-Vertex (Claude tool use)", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"What's the weather in Bangalore?\"}],\n \"tools\": [{\n \"type\": \"function\",\n \"function\": {\n \"name\": \"get_weather\",\n \"parameters\": {\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}\n }\n }]\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} - } - } - ] - } - ] - }, - { - "name": "11. Cross-Provider Feature Tests", - "description": "Same feature exercised across multiple providers via Bifrost's unified routing. Each sub-folder is one capability; each request differs only by `model` (or path) to show that Bifrost translates the request shape per-provider.", - "item": [ - { - "name": "Structured Output cross-cut", - "item": [ + "messages" + ] + } + } + }, { - "name": "openai/gpt-4o-mini (json_schema)", - "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], + "name": "Drop-in /anthropic: claude-sonnet-4-6 vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\n \"type\": \"json_schema\",\n \"json_schema\": {\n \"name\": \"city\",\n \"strict\": true,\n \"schema\": {\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}\n }\n }\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } } }, { - "name": "anthropic/claude-haiku (forced tool)", - "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Forced tool: emit_city invoked with schema-compliant arguments', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('emit_city'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); pm.expect(a).to.have.property('country').that.is.a('string'); pm.expect(a).to.have.property('pop').that.is.a('number'); }); }"]}}], + "name": "Drop-in /anthropic: claude-haiku-4-5 vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"tools\": [{\n \"type\": \"function\",\n \"function\": {\n \"name\": \"emit_city\",\n \"parameters\": {\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"]}\n }\n }],\n \"tool_choice\": {\"type\":\"function\",\"function\":{\"name\":\"emit_city\"}}\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } } }, { - "name": "gemini/gemini-2.5-flash (responseSchema)", - "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (pp) { return pp && pp.text; }); var c = t ? t.text : ''; pm.expect(c, 'candidates[0].content.parts[*].text empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('parts.text not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], + "name": "Drop-in /genai: gemini-2.5-flash vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Gemini vision: text part describes image', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (p) { return p && p.text; }); pm.expect(t, 'no text part').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Extract city/country/pop for Paris.\"}]}],\n \"generationConfig\": {\n \"responseMimeType\": \"application/json\",\n \"responseSchema\": {\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"},\"country\":{\"type\":\"STRING\"},\"pop\":{\"type\":\"NUMBER\"}}}\n }\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/gemini-2.5-flash:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","gemini-2.5-flash:generateContent"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/gemini-2.5-flash:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "gemini-2.5-flash:generateContent" + ] + } } }, { - "name": "vertex/gemini-2.5-pro (responseSchema)", - "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (pp) { return pp && pp.text; }); var c = t ? t.text : ''; pm.expect(c, 'candidates[0].content.parts[*].text empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('parts.text not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], + "name": "Drop-in /genai: gemini-2.5-pro vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Gemini vision: text part describes image', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (p) { return p && p.text; }); pm.expect(t, 'no text part').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Extract city/country/pop for Paris.\"}]}],\n \"generationConfig\": {\n \"responseMimeType\": \"application/json\",\n \"responseSchema\": {\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"},\"country\":{\"type\":\"STRING\"},\"pop\":{\"type\":\"NUMBER\"}}}\n }\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/gemini-2.5-pro:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","gemini-2.5-pro:generateContent"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/gemini-2.5-pro:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "gemini-2.5-pro:generateContent" + ] + } } } ] }, { - "name": "Web Search cross-cut", + "name": "Cross-Cut Round 26: Drop-in Streaming Matrix (native shapes)", "item": [ { - "name": "openai/gpt-4o (web_search_preview)", + "name": "Drop-in /openai stream: gpt-4o", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o\",\n \"input\": \"Latest news on AI regulation in EU.\",\n \"tools\": [{ \"type\": \"web_search_preview\" }]\n}"}, - "url": {"raw":"{{baseUrl}}/v1/responses","host":["{{baseUrl}}"],"path":["v1","responses"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } } }, { - "name": "anthropic/claude-opus-4-7 (web_search_20250305)", + "name": "Drop-in /openai stream: gpt-4o-mini", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-version","value":"2023-06-01"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news on AI regulation in EU.\"}],\n \"tools\": [{ \"type\": \"web_search_20250305\", \"name\": \"web_search\", \"max_uses\": 3 }]\n}"}, - "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } } }, { - "name": "gemini (googleSearch)", + "name": "Drop-in /openai stream: gpt-5", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Latest news on AI regulation in EU.\"}]}],\n \"tools\": [{ \"googleSearch\": {} }]\n}"}, - "url": {"raw":"{{baseUrl}}/genai/v1beta/models/gemini-2.5-flash:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","gemini-2.5-flash:generateContent"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "chat", + "completions" + ] + } } - } - ] - }, - { - "name": "Function Calling cross-cut", - "item": [ + }, { - "name": "openai/gpt-4o-mini", - "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], + "name": "Drop-in /anthropic stream: claude-opus-4-7", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } } }, { - "name": "anthropic/claude-haiku-4-5", - "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], + "name": "Drop-in /anthropic stream: claude-sonnet-4-6", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } } }, { - "name": "bedrock/global.anthropic.claude-sonnet-4-6", - "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], + "name": "Drop-in /anthropic stream: claude-haiku-4-5", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } } }, { - "name": "anthropic/claude-sonnet-5", - "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], + "name": "Drop-in /bedrock stream: claude-haiku Converse", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Bedrock stream: AWS event-stream or SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected stream content-type, got ' + ct).to.match(/event-stream|vnd\\.amazon\\.eventstream/); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Count 1-5.\"}]}],\n \"inferenceConfig\": {\"maxTokens\": 256}\n}" + }, + "url": { + "raw": "{{baseUrl}}/bedrock/model/global.anthropic.claude-haiku-4-5-20251001-v1:0/converse-stream", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "bedrock", + "model", + "global.anthropic.claude-haiku-4-5-20251001-v1:0", + "converse-stream" + ] + } } }, { - "name": "bedrock/global.anthropic.claude-sonnet-5", - "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], + "name": "Drop-in /genai stream: gemini-2.5-flash", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Count 1-5.\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/gemini-2.5-flash:streamGenerateContent?alt=sse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "gemini-2.5-flash:streamGenerateContent" + ], + "query": [ + { + "key": "alt", + "value": "sse" + } + ] + } } }, { - "name": "gemini/gemini-2.5-flash", - "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], + "name": "Drop-in /genai stream: gemini-2.5-pro", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Count 1-5.\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/genai/v1beta/models/gemini-2.5-pro:streamGenerateContent?alt=sse", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "genai", + "v1beta", + "models", + "gemini-2.5-pro:streamGenerateContent" + ], + "query": [ + { + "key": "alt", + "value": "sse" + } + ] + } } } ] }, { - "name": "Streaming cross-cut", + "name": "Cross-Cut Round 27: Drop-in Umbrella Vision Matrix (vision via /langchain, /litellm, /pydanticai, /cursor)", "item": [ { - "name": "openai/gpt-4o-mini", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} - } - }, - { - "name": "anthropic/claude-haiku-4-5", + "name": "Drop-in /langchain vision: OpenAI shape", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/langchain/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "langchain", + "v1", + "chat", + "completions" + ] + } } }, { - "name": "bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0", + "name": "Drop-in /langchain vision: Anthropic shape", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/langchain/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "langchain", + "v1", + "messages" + ] + } } }, { - "name": "anthropic/claude-sonnet-5", + "name": "Drop-in /langchain vision: Gemini shape", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: text part describes image', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (p) { return p && p.text; }); pm.expect(t, 'no text part').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/langchain/v1beta/models/gemini-2.5-flash:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "langchain", + "v1beta", + "models", + "gemini-2.5-flash:generateContent" + ] + } } }, { - "name": "bedrock/global.anthropic.claude-sonnet-5", + "name": "Drop-in /litellm vision: OpenAI shape", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/litellm/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "litellm", + "v1", + "chat", + "completions" + ] + } } }, { - "name": "gemini/gemini-2.5-flash", - "request": { - "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} - } - } - ] - }, - { - "name": "Vision cross-cut", - "item": [ - { - "name": "openai/gpt-4o-mini", + "name": "Drop-in /litellm vision: Anthropic shape", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"What's in this image?\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/litellm/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "litellm", + "v1", + "messages" + ] + } } }, { - "name": "anthropic/claude-haiku-4-5", + "name": "Drop-in /litellm vision: Gemini shape", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: text part describes image', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (p) { return p && p.text; }); pm.expect(t, 'no text part').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"What's in this image?\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/litellm/v1beta/models/gemini-2.5-flash:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "litellm", + "v1beta", + "models", + "gemini-2.5-flash:generateContent" + ] + } } }, { - "name": "gemini/gemini-2.5-flash", + "name": "Drop-in /pydanticai vision: OpenAI shape", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"What's in this image?\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, - "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/pydanticai/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "pydanticai", + "v1", + "chat", + "completions" + ] + } } - } - ] - }, - { - "name": "Context Compaction cross-cut", - "item": [ + }, { - "name": "Compaction via native Bifrost API (openai/gpt-4o)", - "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Compaction: object is response.compaction', function () { var j = pm.response.json(); pm.expect(j.object).to.equal('response.compaction'); }); pm.test('Compaction: output is non-empty array', function () { var j = pm.response.json(); pm.expect(j.output).to.be.an('array').and.not.empty; }); pm.test('Compaction: last output item has type response.compaction and encrypted_content', function () { var j = pm.response.json(); var last = j.output[j.output.length - 1]; pm.expect(last.type).to.equal('compaction'); pm.expect(last.encrypted_content).to.be.a('string').and.not.empty; }); pm.test('Compaction: usage is present', function () { var j = pm.response.json(); pm.expect(j.usage).to.be.an('object'); pm.expect(j.usage.input_tokens).to.be.a('number'); }); }"]}}], + "name": "Drop-in /pydanticai vision: Anthropic shape", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o\",\n \"input\": [\n {\"role\": \"user\", \"content\": \"What is the capital of France?\"},\n {\"role\": \"assistant\", \"content\": \"The capital of France is Paris.\"},\n {\"role\": \"user\", \"content\": \"What is the population of Paris?\"},\n {\"role\": \"assistant\", \"content\": \"Paris has a population of approximately 2.1 million in the city proper, and around 12 million in the greater metropolitan area.\"},\n {\"role\": \"user\", \"content\": \"What is Paris known for?\"},\n {\"role\": \"assistant\", \"content\": \"Paris is known for the Eiffel Tower, the Louvre Museum, Notre-Dame Cathedral, world-class cuisine, fashion, and its rich history as a cultural and political center of Europe.\"}\n ]\n}"}, - "url": {"raw":"{{baseUrl}}/v1/responses/compact","host":["{{baseUrl}}"],"path":["v1","responses","compact"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/pydanticai/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "pydanticai", + "v1", + "messages" + ] + } } }, { - "name": "Compaction (OpenAI gpt-4o)", - "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Compaction: object is response.compaction', function () { var j = pm.response.json(); pm.expect(j.object).to.equal('response.compaction'); }); pm.test('Compaction: output is non-empty array', function () { var j = pm.response.json(); pm.expect(j.output).to.be.an('array').and.not.empty; }); pm.test('Compaction: last output item has type response.compaction and encrypted_content', function () { var j = pm.response.json(); var last = j.output[j.output.length - 1]; pm.expect(last.type).to.equal('compaction'); pm.expect(last.encrypted_content).to.be.a('string').and.not.empty; }); pm.test('Compaction: usage is present', function () { var j = pm.response.json(); pm.expect(j.usage).to.be.an('object'); pm.expect(j.usage.input_tokens).to.be.a('number'); }); }"]}}], + "name": "Drop-in /pydanticai vision: Gemini shape", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: text part describes image', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (p) { return p && p.text; }); pm.expect(t, 'no text part').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"input\": [\n {\"role\": \"user\", \"content\": \"What is the capital of France?\"},\n {\"role\": \"assistant\", \"content\": \"The capital of France is Paris.\"},\n {\"role\": \"user\", \"content\": \"What is the population of Paris?\"},\n {\"role\": \"assistant\", \"content\": \"Paris has a population of approximately 2.1 million in the city proper, and around 12 million in the greater metropolitan area.\"},\n {\"role\": \"user\", \"content\": \"What is Paris known for?\"},\n {\"role\": \"assistant\", \"content\": \"Paris is known for the Eiffel Tower, the Louvre Museum, Notre-Dame Cathedral, world-class cuisine, fashion, and its rich history as a cultural and political center of Europe.\"}\n ]\n}"}, - "url": {"raw":"{{baseUrl}}/openai/v1/responses/compact","host":["{{baseUrl}}"],"path":["openai","v1","responses","compact"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/pydanticai/v1beta/models/gemini-2.5-flash:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "pydanticai", + "v1beta", + "models", + "gemini-2.5-flash:generateContent" + ] + } } }, { - "name": "Compaction (Azure gpt-4o)", - "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Compaction: object is response.compaction', function () { var j = pm.response.json(); pm.expect(j.object).to.equal('response.compaction'); }); pm.test('Compaction: output is non-empty array', function () { var j = pm.response.json(); pm.expect(j.output).to.be.an('array').and.not.empty; }); pm.test('Compaction: last output item has type response.compaction and encrypted_content', function () { var j = pm.response.json(); var last = j.output[j.output.length - 1]; pm.expect(last.type).to.equal('compaction'); pm.expect(last.encrypted_content).to.be.a('string').and.not.empty; }); pm.test('Compaction: usage is present', function () { var j = pm.response.json(); pm.expect(j.usage).to.be.an('object'); pm.expect(j.usage.input_tokens).to.be.a('number'); }); }"]}}], + "name": "Drop-in /cursor vision: OpenAI shape", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"azure/{{azureDeployment}}\",\n \"input\": [\n {\"role\": \"user\", \"content\": \"What is the capital of France?\"},\n {\"role\": \"assistant\", \"content\": \"The capital of France is Paris.\"},\n {\"role\": \"user\", \"content\": \"What is the population of Paris?\"},\n {\"role\": \"assistant\", \"content\": \"Paris has a population of approximately 2.1 million in the city proper, and around 12 million in the greater metropolitan area.\"},\n {\"role\": \"user\", \"content\": \"What is Paris known for?\"},\n {\"role\": \"assistant\", \"content\": \"Paris is known for the Eiffel Tower, the Louvre Museum, Notre-Dame Cathedral, world-class cuisine, fashion, and its rich history as a cultural and political center of Europe.\"}\n ]\n}"}, - "url": {"raw":"{{baseUrl}}/openai/v1/responses/compact","host":["{{baseUrl}}"],"path":["openai","v1","responses","compact"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/cursor/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "cursor", + "v1", + "chat", + "completions" + ] + } } }, { - "name": "Compaction (xAI grok-4.3)", - "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Compaction: object is response.compaction', function () { var j = pm.response.json(); pm.expect(j.object).to.equal('response.compaction'); }); pm.test('Compaction: output is non-empty array', function () { var j = pm.response.json(); pm.expect(j.output).to.be.an('array').and.not.empty; }); pm.test('Compaction: last output item has type response.compaction and encrypted_content', function () { var j = pm.response.json(); var last = j.output[j.output.length - 1]; pm.expect(last.type).to.equal('compaction'); pm.expect(last.encrypted_content).to.be.a('string').and.not.empty; }); pm.test('Compaction: usage is present', function () { var j = pm.response.json(); pm.expect(j.usage).to.be.an('object'); pm.expect(j.usage.input_tokens).to.be.a('number'); }); }"]}}], + "name": "Drop-in /cursor vision: Anthropic shape", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"xai/grok-4.3\",\n \"input\": [\n {\"role\": \"user\", \"content\": \"What is the capital of France?\"},\n {\"role\": \"assistant\", \"content\": \"The capital of France is Paris.\"},\n {\"role\": \"user\", \"content\": \"What is the population of Paris?\"},\n {\"role\": \"assistant\", \"content\": \"Paris has a population of approximately 2.1 million in the city proper, and around 12 million in the greater metropolitan area.\"},\n {\"role\": \"user\", \"content\": \"What is Paris known for?\"},\n {\"role\": \"assistant\", \"content\": \"Paris is known for the Eiffel Tower, the Louvre Museum, Notre-Dame Cathedral, world-class cuisine, fashion, and its rich history as a cultural and political center of Europe.\"}\n ]\n}"}, - "url": {"raw":"{{baseUrl}}/v1/responses/compact","host":["{{baseUrl}}"],"path":["v1","responses","compact"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/cursor/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "cursor", + "v1", + "messages" + ] + } } }, { - "name": "Compaction with instructions (OpenAI gpt-4o)", - "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Compaction: object is response.compaction', function () { var j = pm.response.json(); pm.expect(j.object).to.equal('response.compaction'); }); pm.test('Compaction: output is non-empty array', function () { var j = pm.response.json(); pm.expect(j.output).to.be.an('array').and.not.empty; }); pm.test('Compaction: last output item has type response.compaction and encrypted_content', function () { var j = pm.response.json(); var last = j.output[j.output.length - 1]; pm.expect(last.type).to.equal('compaction'); pm.expect(last.encrypted_content).to.be.a('string').and.not.empty; }); pm.test('Compaction: usage is present', function () { var j = pm.response.json(); pm.expect(j.usage).to.be.an('object'); pm.expect(j.usage.input_tokens).to.be.a('number'); }); }"]}}], + "name": "Drop-in /cursor vision: Gemini shape", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: text part describes image', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (p) { return p && p.text; }); pm.expect(t, 'no text part').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }" + ] + } + } + ], "request": { "method": "POST", - "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], - "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"input\": [\n {\"role\": \"user\", \"content\": \"What is the capital of France?\"},\n {\"role\": \"assistant\", \"content\": \"The capital of France is Paris.\"},\n {\"role\": \"user\", \"content\": \"What is the population of Paris?\"},\n {\"role\": \"assistant\", \"content\": \"Paris has a population of approximately 2.1 million in the city proper, and around 12 million in the greater metropolitan area.\"},\n {\"role\": \"user\", \"content\": \"What is Paris known for?\"},\n {\"role\": \"assistant\", \"content\": \"Paris is known for the Eiffel Tower, the Louvre Museum, Notre-Dame Cathedral, world-class cuisine, fashion, and its rich history as a cultural and political center of Europe.\"}\n ],\n \"instructions\": \"You are a helpful geography assistant. Be concise.\"\n}"}, - "url": {"raw":"{{baseUrl}}/openai/v1/responses/compact","host":["{{baseUrl}}"],"path":["openai","v1","responses","compact"]} + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/cursor/v1beta/models/gemini-2.5-flash:generateContent", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "cursor", + "v1beta", + "models", + "gemini-2.5-flash:generateContent" + ] + } } } ] }, { - "name": "MCP Tool Handling cross-cut", - "description": "Regression #3795: a provider-side `type:\"mcp\"` server tool in a Responses request must be silently dropped for providers without an MCP connector (Bedrock, Vertex), not rejected. Function tools survive the strip. Function-bearing rows force the surviving function tool via tool_choice and verify it is invoked (output function_call / streamed response.function_call_arguments). Lone-mcp rows exercise the all-tools-dropped edge case: zero tools left after the strip must still return a text answer with NO tool call. Server/MCP tool calls are NOT expected for Bedrock/Vertex Claude — that is correct. Covers native + /openai drop-in + streaming.", + "name": "Cross-Cut Round 28: Passthrough Advanced Matrix (features via *_passthrough byte-for-byte routes)", "item": [ { - "name": "bedrock/global.anthropic.claude-opus-4-7 · lone server-mcp dropped", + "name": "Passthrough /openai: structured output (json_schema)", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795 'all tools dropped' edge case: the mcp server tool is the ONLY tool, so after", - "// the strip there are zero tools left. The request must still return a normal text", - "// completion, and NO tool call may appear (the mcp tool was removed, not made callable).", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') === -1) {", - " pm.test('text answer returned, no tool call (mcp dropped, zero tools left)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'no tool call expected (mcp dropped, no function tools), got: ' + JSON.stringify(out).slice(0, 200)).to.equal(0);", - " var txt = (typeof j.output_text === 'string') ? j.output_text : '';", - " var hasText = txt.length > 0 ||", - " out.some(function (o) { return o && o.type === 'message' && o.content && o.content.length; }) ||", - " (j.output && j.output.message && j.output.message.content && j.output.message.content.length);", - " pm.expect(hasText, 'expected a text/message answer, got: ' + JSON.stringify(j).slice(0, 200)).to.be.true;", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Structured output: JSON with city/country/pop', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('not JSON: ' + e.message); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" ] } } @@ -2555,61 +31072,39 @@ { "key": "Content-Type", "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"What is 2+2? Answer in one word.\",\n \"tools\": [\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ]\n}" + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/openai_passthrough/v1/chat/completions", "host": [ "{{baseUrl}}" ], "path": [ + "openai_passthrough", "v1", - "responses" + "chat", + "completions" ] } } }, { - "name": "bedrock/global.anthropic.claude-opus-4-7 · server-mcp + function kept (forced call)", + "name": "Passthrough /openai: function calling", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", - "// so a server/MCP tool call must NOT happen here — that is expected and fine.", - "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "var EXPECT = 'get_weather';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') !== -1) {", - " pm.test('streaming response invoked function tool ' + EXPECT, function () {", - " pm.expect(raw, 'no function_call events in SSE stream')", - " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", - " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", - " });", - "} else {", - " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", - " var names = calls.map(function (o) { return o.name; });", - " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Function call: get_weather invoked', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls && j.choices[0].message.tool_calls[0]; pm.expect(tc, 'no tool_calls').to.be.ok; pm.expect(tc.function.name).to.equal('get_weather'); }); }" ] } } @@ -2620,61 +31115,39 @@ { "key": "Content-Type", "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/openai_passthrough/v1/chat/completions", "host": [ "{{baseUrl}}" ], "path": [ + "openai_passthrough", "v1", - "responses" + "chat", + "completions" ] } } }, { - "name": "bedrock/global.anthropic.claude-opus-4-7 · 2 server-mcp + 2 function (#3795 shape, forced call)", + "name": "Passthrough /openai: streaming", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", - "// so a server/MCP tool call must NOT happen here — that is expected and fine.", - "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "var EXPECT = 'set_serialviewer_pro_query';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') !== -1) {", - " pm.test('streaming response invoked function tool ' + EXPECT, function () {", - " pm.expect(raw, 'no function_call events in SSE stream')", - " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", - " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", - " });", - "} else {", - " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", - " var names = calls.map(function (o) { return o.name; });", - " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }" ] } } @@ -2685,57 +31158,85 @@ { "key": "Content-Type", "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"Build a MongoDB find filter for the most common error codes in the orders collection over the last month.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\",\n \"description\": \"Writes a MongoDB find filter.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"filter\": {\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"filter\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"time\",\n \"server_url\": \"https://bifrost.invalid/mcp/time\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"function\",\n \"name\": \"mongodb-explain\",\n \"description\": \"Returns query-plan statistics.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"collection\": {\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"additionalProperties\": false\n }\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\"\n }\n}" + "raw": "{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/openai_passthrough/v1/chat/completions", "host": [ "{{baseUrl}}" ], "path": [ + "openai_passthrough", "v1", - "responses" + "chat", + "completions" ] } } }, { - "name": "bedrock/global.anthropic.claude-sonnet-4-6 · lone server-mcp dropped", + "name": "Passthrough /anthropic: function calling (tool_use)", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795 'all tools dropped' edge case: the mcp server tool is the ONLY tool, so after", - "// the strip there are zero tools left. The request must still return a normal text", - "// completion, and NO tool call may appear (the mcp tool was removed, not made callable).", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') === -1) {", - " pm.test('text answer returned, no tool call (mcp dropped, zero tools left)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'no tool call expected (mcp dropped, no function tools), got: ' + JSON.stringify(out).slice(0, 200)).to.equal(0);", - " var txt = (typeof j.output_text === 'string') ? j.output_text : '';", - " var hasText = txt.length > 0 ||", - " out.some(function (o) { return o && o.type === 'message' && o.content && o.content.length; }) ||", - " (j.output && j.output.message && j.output.message.content && j.output.message.content.length);", - " pm.expect(hasText, 'expected a text/message answer, got: ' + JSON.stringify(j).slice(0, 200)).to.be.true;", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Anthropic: tool_use block', function () { var j = pm.response.json(); var tu = (j.content || []).find(function (b) { return b.type === 'tool_use'; }); pm.expect(tu, 'no tool_use').to.be.ok; pm.expect(tu.name).to.equal('get_weather'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"name\":\"get_weather\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic_passthrough/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic_passthrough", + "v1", + "messages" + ] + } + } + }, + { + "name": "Passthrough /anthropic: vision", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }" ] } } @@ -2746,61 +31247,42 @@ { "key": "Content-Type", "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"input\": \"What is 2+2? Answer in one word.\",\n \"tools\": [\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ]\n}" + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/anthropic_passthrough/v1/messages", "host": [ "{{baseUrl}}" ], "path": [ + "anthropic_passthrough", "v1", - "responses" + "messages" ] } } }, { - "name": "bedrock/global.anthropic.claude-sonnet-4-6 · server-mcp + function kept (forced call)", + "name": "Passthrough /anthropic: multi-turn", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", - "// so a server/MCP tool call must NOT happen here — that is expected and fine.", - "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "var EXPECT = 'get_weather';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') !== -1) {", - " pm.test('streaming response invoked function tool ' + EXPECT, function () {", - " pm.expect(raw, 'no function_call events in SSE stream')", - " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", - " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", - " });", - "} else {", - " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", - " var names = calls.map(function (o) { return o.name; });", - " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Multi-turn: response present', function () { var j = pm.response.json(); var c = (j.content && j.content.length > 0) || (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content); pm.expect(c, 'no content').to.be.ok; }); }" ] } } @@ -2811,61 +31293,42 @@ { "key": "Content-Type", "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/anthropic_passthrough/v1/messages", "host": [ "{{baseUrl}}" ], "path": [ + "anthropic_passthrough", "v1", - "responses" + "messages" ] } } }, { - "name": "bedrock/global.anthropic.claude-sonnet-4-6 · 2 server-mcp + 2 function (#3795 shape, forced call)", + "name": "Passthrough /anthropic: streaming", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", - "// so a server/MCP tool call must NOT happen here — that is expected and fine.", - "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "var EXPECT = 'set_serialviewer_pro_query';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') !== -1) {", - " pm.test('streaming response invoked function tool ' + EXPECT, function () {", - " pm.expect(raw, 'no function_call events in SSE stream')", - " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", - " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", - " });", - "} else {", - " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", - " var names = calls.map(function (o) { return o.name; });", - " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }" ] } } @@ -2876,57 +31339,42 @@ { "key": "Content-Type", "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"input\": \"Build a MongoDB find filter for the most common error codes in the orders collection over the last month.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\",\n \"description\": \"Writes a MongoDB find filter.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"filter\": {\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"filter\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"time\",\n \"server_url\": \"https://bifrost.invalid/mcp/time\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"function\",\n \"name\": \"mongodb-explain\",\n \"description\": \"Returns query-plan statistics.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"collection\": {\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"additionalProperties\": false\n }\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\"\n }\n}" + "raw": "{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/anthropic_passthrough/v1/messages", "host": [ "{{baseUrl}}" ], "path": [ + "anthropic_passthrough", "v1", - "responses" + "messages" ] } } }, { - "name": "vertex/claude-opus-4-7 · lone server-mcp dropped", + "name": "Passthrough /genai: function calling", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795 'all tools dropped' edge case: the mcp server tool is the ONLY tool, so after", - "// the strip there are zero tools left. The request must still return a normal text", - "// completion, and NO tool call may appear (the mcp tool was removed, not made callable).", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') === -1) {", - " pm.test('text answer returned, no tool call (mcp dropped, zero tools left)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'no tool call expected (mcp dropped, no function tools), got: ' + JSON.stringify(out).slice(0, 200)).to.equal(0);", - " var txt = (typeof j.output_text === 'string') ? j.output_text : '';", - " var hasText = txt.length > 0 ||", - " out.some(function (o) { return o && o.type === 'message' && o.content && o.content.length; }) ||", - " (j.output && j.output.message && j.output.message.content && j.output.message.content.length);", - " pm.expect(hasText, 'expected a text/message answer, got: ' + JSON.stringify(j).slice(0, 200)).to.be.true;", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Gemini: functionCall present', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var fc = parts.find(function (p) { return p && p.functionCall; }); pm.expect(fc, 'no functionCall').to.be.ok; pm.expect(fc.functionCall.name).to.equal('get_weather'); }); }" ] } } @@ -2937,61 +31385,39 @@ { "key": "Content-Type", "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"input\": \"What is 2+2? Answer in one word.\",\n \"tools\": [\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ]\n}" + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Weather in Lagos, Nigeria?\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"get_weather\",\"parameters\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"}},\"required\":[\"city\"]}}]}]\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/genai_passthrough/v1beta/models/gemini-2.5-flash:generateContent", "host": [ "{{baseUrl}}" ], "path": [ - "v1", - "responses" + "genai_passthrough", + "v1beta", + "models", + "gemini-2.5-flash:generateContent" ] } } }, { - "name": "vertex/claude-opus-4-7 · server-mcp + function kept (forced call)", + "name": "Passthrough /genai: vision", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", - "// so a server/MCP tool call must NOT happen here — that is expected and fine.", - "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "var EXPECT = 'get_weather';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') !== -1) {", - " pm.test('streaming response invoked function tool ' + EXPECT, function () {", - " pm.expect(raw, 'no function_call events in SSE stream')", - " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", - " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", - " });", - "} else {", - " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", - " var names = calls.map(function (o) { return o.name; });", - " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Vision: text part describes image', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (p) { return p && p.text; }); pm.expect(t, 'no text part').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }" ] } } @@ -3002,61 +31428,39 @@ { "key": "Content-Type", "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/genai_passthrough/v1beta/models/gemini-2.5-flash:generateContent", "host": [ "{{baseUrl}}" ], "path": [ - "v1", - "responses" + "genai_passthrough", + "v1beta", + "models", + "gemini-2.5-flash:generateContent" ] } } }, { - "name": "vertex/claude-opus-4-7 · 2 server-mcp + 2 function (#3795 shape, forced call)", + "name": "Passthrough /genai: streaming", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", - "// so a server/MCP tool call must NOT happen here — that is expected and fine.", - "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "var EXPECT = 'set_serialviewer_pro_query';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') !== -1) {", - " pm.test('streaming response invoked function tool ' + EXPECT, function () {", - " pm.expect(raw, 'no function_call events in SSE stream')", - " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", - " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", - " });", - "} else {", - " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", - " var names = calls.map(function (o) { return o.name; });", - " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }" ] } } @@ -3067,57 +31471,45 @@ { "key": "Content-Type", "value": "application/json" + }, + { + "key": "x-goog-api-key", + "value": "{{genaiKey}}" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"input\": \"Build a MongoDB find filter for the most common error codes in the orders collection over the last month.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\",\n \"description\": \"Writes a MongoDB find filter.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"filter\": {\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"filter\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"time\",\n \"server_url\": \"https://bifrost.invalid/mcp/time\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"function\",\n \"name\": \"mongodb-explain\",\n \"description\": \"Returns query-plan statistics.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"collection\": {\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"additionalProperties\": false\n }\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\"\n }\n}" + "raw": "{\n \"contents\": [{\"parts\":[{\"text\":\"Count 1-5.\"}]}]\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/genai_passthrough/v1beta/models/gemini-2.5-flash:streamGenerateContent?alt=sse", "host": [ "{{baseUrl}}" ], "path": [ - "v1", - "responses" + "genai_passthrough", + "v1beta", + "models", + "gemini-2.5-flash:streamGenerateContent" + ], + "query": [ + { + "key": "alt", + "value": "sse" + } ] } } }, { - "name": "vertex/claude-sonnet-4-6 · lone server-mcp dropped", + "name": "Passthrough /azure: structured output (json_schema)", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795 'all tools dropped' edge case: the mcp server tool is the ONLY tool, so after", - "// the strip there are zero tools left. The request must still return a normal text", - "// completion, and NO tool call may appear (the mcp tool was removed, not made callable).", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') === -1) {", - " pm.test('text answer returned, no tool call (mcp dropped, zero tools left)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'no tool call expected (mcp dropped, no function tools), got: ' + JSON.stringify(out).slice(0, 200)).to.equal(0);", - " var txt = (typeof j.output_text === 'string') ? j.output_text : '';", - " var hasText = txt.length > 0 ||", - " out.some(function (o) { return o && o.type === 'message' && o.content && o.content.length; }) ||", - " (j.output && j.output.message && j.output.message.content && j.output.message.content.length);", - " pm.expect(hasText, 'expected a text/message answer, got: ' + JSON.stringify(j).slice(0, 200)).to.be.true;", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Structured output: JSON with city/country/pop', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('not JSON: ' + e.message); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }" ] } } @@ -3128,61 +31520,47 @@ { "key": "Content-Type", "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"input\": \"What is 2+2? Answer in one word.\",\n \"tools\": [\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ]\n}" + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/azure_passthrough/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", "host": [ "{{baseUrl}}" ], "path": [ - "v1", - "responses" + "azure_passthrough", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } ] } } }, { - "name": "vertex/claude-sonnet-4-6 · server-mcp + function kept (forced call)", + "name": "Passthrough /azure: vision", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", - "// so a server/MCP tool call must NOT happen here — that is expected and fine.", - "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "var EXPECT = 'get_weather';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') !== -1) {", - " pm.test('streaming response invoked function tool ' + EXPECT, function () {", - " pm.expect(raw, 'no function_call events in SSE stream')", - " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", - " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", - " });", - "} else {", - " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", - " var names = calls.map(function (o) { return o.name; });", - " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }" ] } } @@ -3193,61 +31571,47 @@ { "key": "Content-Type", "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/azure_passthrough/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", "host": [ "{{baseUrl}}" ], "path": [ - "v1", - "responses" + "azure_passthrough", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } ] } } }, { - "name": "vertex/claude-sonnet-4-6 · 2 server-mcp + 2 function (#3795 shape, forced call)", + "name": "Passthrough /azure: streaming", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", - "// so a server/MCP tool call must NOT happen here — that is expected and fine.", - "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "var EXPECT = 'set_serialviewer_pro_query';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') !== -1) {", - " pm.test('streaming response invoked function tool ' + EXPECT, function () {", - " pm.expect(raw, 'no function_call events in SSE stream')", - " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", - " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", - " });", - "} else {", - " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", - " var names = calls.map(function (o) { return o.name; });", - " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }" ] } } @@ -3258,61 +31622,53 @@ { "key": "Content-Type", "value": "application/json" + }, + { + "key": "api-key", + "value": "{{openaiKey}}" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"input\": \"Build a MongoDB find filter for the most common error codes in the orders collection over the last month.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\",\n \"description\": \"Writes a MongoDB find filter.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"filter\": {\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"filter\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"time\",\n \"server_url\": \"https://bifrost.invalid/mcp/time\",\n \"require_approval\": \"never\"\n },\n {\n \"type\": \"function\",\n \"name\": \"mongodb-explain\",\n \"description\": \"Returns query-plan statistics.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"collection\": {\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"additionalProperties\": false\n }\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"set_serialviewer_pro_query\"\n }\n}" + "raw": "{\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/azure_passthrough/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}", "host": [ "{{baseUrl}}" ], "path": [ - "v1", - "responses" + "azure_passthrough", + "openai", + "deployments", + "{{azureDeployment}}", + "chat", + "completions" + ], + "query": [ + { + "key": "api-version", + "value": "{{azureApiVersion}}" + } ] } } - }, + } + ] + }, + { + "name": "Cross-Cut Round 29: Mid-Conversation System Message Matrix", + "description": "Verifies mid-conversation role:system handling across providers.\n- Anthropic + Opus 4.8: system emitted as role:\"system\" in messages array (native support, no beta header).\n- Bedrock + Vertex + Opus 4.7: system content merged into top-level system field (fallback — positional semantics lost, data preserved).\nPlacement rule: system must end the array OR be immediately followed by an assistant turn.", + "item": [ { - "name": "bedrock/global.anthropic.claude-opus-4-7 · /openai drop-in · forced function call", + "name": "Cross-cut: anthropic/claude-opus-4-8 mid-conv system (ends array)", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", - "// so a server/MCP tool call must NOT happen here — that is expected and fine.", - "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "var EXPECT = 'get_weather';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') !== -1) {", - " pm.test('streaming response invoked function tool ' + EXPECT, function () {", - " pm.expect(raw, 'no function_call events in SSE stream')", - " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", - " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", - " });", - "} else {", - " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", - " var names = calls.map(function (o) { return o.name; });", - " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" ] } } @@ -3327,58 +31683,30 @@ ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + "raw": "{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"},\n {\"role\":\"system\",\"content\":\"Respond only in one word.\"}\n ]\n}" }, "url": { - "raw": "{{baseUrl}}/openai/v1/responses", + "raw": "{{baseUrl}}/v1/chat/completions", "host": [ "{{baseUrl}}" ], "path": [ - "openai", "v1", - "responses" + "chat", + "completions" ] } } }, { - "name": "vertex/claude-opus-4-7 · /openai drop-in · forced function call", + "name": "Cross-cut: anthropic/claude-opus-4-8 mid-conv system (before assistant)", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", - "// so a server/MCP tool call must NOT happen here — that is expected and fine.", - "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "var EXPECT = 'get_weather';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') !== -1) {", - " pm.test('streaming response invoked function tool ' + EXPECT, function () {", - " pm.expect(raw, 'no function_call events in SSE stream')", - " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", - " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", - " });", - "} else {", - " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", - " var names = calls.map(function (o) { return o.name; });", - " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" ] } } @@ -3393,58 +31721,30 @@ ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + "raw": "{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"system\",\"content\":\"From now on be very concise.\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"}\n ]\n}" }, "url": { - "raw": "{{baseUrl}}/openai/v1/responses", + "raw": "{{baseUrl}}/v1/chat/completions", "host": [ "{{baseUrl}}" ], "path": [ - "openai", "v1", - "responses" + "chat", + "completions" ] } } }, { - "name": "bedrock/global.anthropic.claude-opus-4-7 · streaming · forced function call", + "name": "Cross-cut: bedrock/claude-opus-4-8 mid-conv system (fallback: merged to top-level)", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", - "// so a server/MCP tool call must NOT happen here — that is expected and fine.", - "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "var EXPECT = 'get_weather';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') !== -1) {", - " pm.test('streaming response invoked function tool ' + EXPECT, function () {", - " pm.expect(raw, 'no function_call events in SSE stream')", - " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", - " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", - " });", - "} else {", - " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", - " var names = calls.map(function (o) { return o.name; });", - " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" ] } } @@ -3459,57 +31759,30 @@ ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n },\n \"stream\": true\n}" + "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"},\n {\"role\":\"system\",\"content\":\"Respond only in one word.\"}\n ]\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/v1/chat/completions", "host": [ "{{baseUrl}}" ], "path": [ "v1", - "responses" + "chat", + "completions" ] } } }, { - "name": "vertex/claude-opus-4-7 · streaming · forced function call", + "name": "Cross-cut: bedrock/claude-opus-4-8 mid-conv system (fallback: merged to top-level)", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// #3795: the mcp SERVER tool is dropped for Bedrock/Vertex Claude (no MCP connector),", - "// so a server/MCP tool call must NOT happen here — that is expected and fine.", - "// We force the surviving FUNCTION tool via tool_choice and verify it is actually called.", - "var ct = (pm.response.headers.get('content-type') || '');", - "var raw = pm.response.text() || '';", - "var EXPECT = 'get_weather';", - "pm.test('mcp server tool dropped, not rejected (#3795)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "if (ct.indexOf('event-stream') !== -1) {", - " pm.test('streaming response invoked function tool ' + EXPECT, function () {", - " pm.expect(raw, 'no function_call events in SSE stream')", - " .to.match(/response\\.function_call_arguments|\"type\"\\s*:\\s*\"function_call\"/);", - " pm.expect(raw, 'expected function ' + EXPECT + ' in stream').to.include(EXPECT);", - " });", - "} else {", - " pm.test('response invoked function tool ' + EXPECT + ' (no server/mcp call)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'expected a function tool call, got: ' + JSON.stringify(j.output || j).slice(0, 220)).to.be.above(0);", - " var names = calls.map(function (o) { return o.name; });", - " pm.expect(names, 'expected ' + EXPECT + ' to be called, got ' + JSON.stringify(names)).to.include(EXPECT);", - " });", - "}" + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" ] } } @@ -3524,48 +31797,30 @@ ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"vertex/claude-opus-4-7\",\n \"input\": \"What's the current weather in Paris? Use the available tool.\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\n \"type\": \"string\",\n \"description\": \"City name\"\n }\n },\n \"required\": [\n \"city\"\n ],\n \"additionalProperties\": false\n }\n },\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n },\n \"stream\": true\n}" + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"},\n {\"role\":\"system\",\"content\":\"Respond only in one word.\"}\n ]\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/v1/chat/completions", "host": [ "{{baseUrl}}" ], "path": [ "v1", - "responses" + "chat", + "completions" ] } } }, - { - "name": "bedrock/global.anthropic.claude-opus-4-7 · tool_choice pins absent fn + lone mcp dropped → reconciled, no 400 (PR #4573)", - "event": [ - { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "// Fix (PR #4573, CodeRabbit): tool_choice pins a function, but the ONLY tool sent is an", - "// unsupported mcp server tool. The mcp tool is dropped -> zero tools, so the dangling", - "// pin must be reconciled away (fall back to Bedrock 'auto') instead of emitting a", - "// toolChoice.tool that references an absent tool, which Bedrock rejects with HTTP 400.", - "// The collection-level 'Status code is 2xx' is the real regression guard (pre-fix => 400).", - "var raw = pm.response.text() || '';", - "pm.test('mcp dropped + dangling tool_choice reconciled, no 400 (PR #4573)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - " pm.expect(raw.toLowerCase(), 'pinned tool must not leak to Bedrock as an unknown tool').to.not.include('tool not found');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "pm.test('no tool call (pin reconciled away, zero tools left)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'expected no tool call (no tools survived), got: ' + JSON.stringify(out).slice(0, 200)).to.equal(0);", - "});" + { + "name": "Cross-cut: vertex/claude-opus-4-8 mid-conv system (fallback: merged to top-level)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" ] } } @@ -3580,48 +31835,30 @@ ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"input\": \"What is 2+2? Answer in one word.\",\n \"tools\": [\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + "raw": "{\n \"model\": \"vertex/claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"},\n {\"role\":\"system\",\"content\":\"Respond only in one word.\"}\n ]\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/v1/chat/completions", "host": [ "{{baseUrl}}" ], "path": [ "v1", - "responses" + "chat", + "completions" ] } } }, { - "name": "bedrock/global.anthropic.claude-sonnet-4-6 · tool_choice pins absent fn + lone mcp dropped → reconciled, no 400 (PR #4573)", + "name": "Cross-cut: anthropic/claude-opus-4-7 mid-conv system (fallback: merged to top-level)", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// Fix (PR #4573, CodeRabbit): tool_choice pins a function, but the ONLY tool sent is an", - "// unsupported mcp server tool. The mcp tool is dropped -> zero tools, so the dangling", - "// pin must be reconciled away (fall back to Bedrock 'auto') instead of emitting a", - "// toolChoice.tool that references an absent tool, which Bedrock rejects with HTTP 400.", - "// The collection-level 'Status code is 2xx' is the real regression guard (pre-fix => 400).", - "var raw = pm.response.text() || '';", - "pm.test('mcp dropped + dangling tool_choice reconciled, no 400 (PR #4573)', function () {", - " pm.expect(raw).to.not.include(\"tool type 'mcp'\");", - " pm.expect(raw.toLowerCase()).to.not.include('is not supported by provider');", - " pm.expect(raw.toLowerCase(), 'pinned tool must not leak to Bedrock as an unknown tool').to.not.include('tool not found');", - "});", - "pm.test('response body is non-empty', function () {", - " pm.expect(raw.length, 'expected a non-empty response body').to.be.above(0);", - "});", - "pm.test('no tool call (pin reconciled away, zero tools left)', function () {", - " var j = pm.response.json();", - " var out = Array.isArray(j.output) ? j.output : [];", - " var calls = out.filter(function (o) { return o && (o.type === 'function_call' || o.type === 'tool_call'); });", - " pm.expect(calls.length, 'expected no tool call (no tools survived), got: ' + JSON.stringify(out).slice(0, 200)).to.equal(0);", - "});" + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" ] } } @@ -3636,1439 +31873,803 @@ ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"input\": \"What is 2+2? Answer in one word.\",\n \"tools\": [\n {\n \"type\": \"mcp\",\n \"server_label\": \"mongodb\",\n \"server_url\": \"https://bifrost.invalid/mcp/mongodb\",\n \"require_approval\": \"never\"\n }\n ],\n \"tool_choice\": {\n \"type\": \"function\",\n \"name\": \"get_weather\"\n }\n}" + "raw": "{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"},\n {\"role\":\"system\",\"content\":\"Respond only in one word.\"}\n ]\n}" }, "url": { - "raw": "{{baseUrl}}/v1/responses", + "raw": "{{baseUrl}}/v1/chat/completions", "host": [ "{{baseUrl}}" ], "path": [ "v1", - "responses" + "chat", + "completions" ] } } - } - ] - } - ] - }, - { - "name": "11b. Vertex GCS Files (/openai + native resumable)", - "description": "Vertex stores files in a customer GCS bucket. CRUD runs via the OpenAI drop-in (storage_config.gcs); resumable upload runs via the native API (mint session -> client PUTs bytes straight to GCS -> cleanup). All [PREVIEW]-tagged: needs a gateway Vertex provider key (server-side) + a GCS bucket. Set vertexGcsBucket and run with --env-var include_preview=1. The OpenAI-drop-in rows send no Authorization header (routing is by provider=vertex).", - "item": [ - { - "name": "[PREVIEW] Vertex: upload file (GCS, /openai)", - "request": { - "method": "POST", - "header": [], - "body": { - "mode": "formdata", - "formdata": [ - { - "key": "file", - "src": "tests/e2e/api/fixtures/sample.jsonl", - "type": "file" - }, - { - "key": "purpose", - "value": "batch", - "type": "text" - }, - { - "key": "provider", - "value": "vertex", - "type": "text" - }, - { - "key": "storage_config[gcs][bucket]", - "value": "{{vertexGcsBucket}}", - "type": "text" - }, - { - "key": "storage_config[gcs][prefix]", - "value": "{{vertexGcsPrefix}}", - "type": "text" - } - ] }, - "url": { - "raw": "{{baseUrl}}/openai/v1/files", - "host": [ - "{{baseUrl}}" - ], - "path": [ - "openai", - "v1", - "files" - ] - } - }, - "event": [ - { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "var j = pm.response.json();", - "pm.test('Vertex upload: gcs backend', function () { pm.expect(j.storage_backend).to.eql('gcs'); });", - "pm.test('Vertex upload: has id', function () { pm.expect(j.id).to.be.a('string').and.not.empty; });", - "pm.collectionVariables.set('vertexFileId', j.id);" - ] - } - } - ] - }, - { - "name": "[PREVIEW] Vertex: list files (GCS, /openai)", - "request": { - "method": "GET", - "header": [], - "url": { - "raw": "{{baseUrl}}/openai/v1/files?provider=vertex&storage_config[gcs][bucket]={{vertexGcsBucket}}&storage_config[gcs][prefix]={{vertexGcsPrefix}}", - "host": [ - "{{baseUrl}}" - ], - "path": [ - "openai", - "v1", - "files" - ], - "query": [ - { - "key": "provider", - "value": "vertex" - }, - { - "key": "storage_config[gcs][bucket]", - "value": "{{vertexGcsBucket}}" - }, - { - "key": "storage_config[gcs][prefix]", - "value": "{{vertexGcsPrefix}}" - } - ] - } - }, - "event": [ - { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "var j = pm.response.json();", - "pm.test('Vertex list: uploaded file present', function () { pm.expect((j.data || []).map(function (f) { return f.id; })).to.include(pm.collectionVariables.get('vertexFileId')); });" - ] - } - } - ] - }, - { - "name": "[PREVIEW] Vertex: retrieve file (GCS, /openai)", - "request": { - "method": "GET", - "header": [], - "url": { - "raw": "{{baseUrl}}/openai/v1/files/{{vertexFileId}}?provider=vertex", - "host": [ - "{{baseUrl}}" - ], - "path": [ - "openai", - "v1", - "files", - "{{vertexFileId}}" - ], - "query": [ - { - "key": "provider", - "value": "vertex" - } - ] - } - }, - "event": [ { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "var j = pm.response.json();", - "pm.test('Vertex retrieve: id matches uploaded', function () { pm.expect(j.id).to.eql(pm.collectionVariables.get('vertexFileId')); });" - ] - } - } - ] - }, - { - "name": "[PREVIEW] Vertex: download file content (GCS, /openai)", - "request": { - "method": "GET", - "header": [], - "url": { - "raw": "{{baseUrl}}/openai/v1/files/{{vertexFileId}}/content?provider=vertex", - "host": [ - "{{baseUrl}}" - ], - "path": [ - "openai", - "v1", - "files", - "{{vertexFileId}}", - "content" - ], - "query": [ + "name": "Cross-cut: anthropic/claude-opus-4-8 mid-conv system drop-in /anthropic (ends array)", + "event": [ { - "key": "provider", - "value": "vertex" + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }" + ] + } } - ] - } - }, - "event": [ - { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "pm.test('Vertex content: non-empty body', function () { pm.expect((pm.response.text() || '').length).to.be.above(0); });" - ] - } - } - ] - }, - { - "name": "[PREVIEW] Vertex: delete file (GCS, /openai)", - "request": { - "method": "DELETE", - "header": [], - "url": { - "raw": "{{baseUrl}}/openai/v1/files/{{vertexFileId}}?provider=vertex", - "host": [ - "{{baseUrl}}" - ], - "path": [ - "openai", - "v1", - "files", - "{{vertexFileId}}" ], - "query": [ - { - "key": "provider", - "value": "vertex" + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 512,\n \"system\": [{\"type\":\"text\",\"text\":\"You are a helpful assistant.\"}],\n \"messages\": [\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Hello\"}]},\n {\"role\":\"assistant\",\"content\":[{\"type\":\"text\",\"text\":\"Hi!\"}]},\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"How are you?\"}]},\n {\"role\":\"system\",\"content\":[{\"type\":\"text\",\"text\":\"Respond only in one word.\"}]}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] } - ] - } - }, - "event": [ - { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "var j = pm.response.json();", - "pm.test('Vertex delete: deleted true', function () { pm.expect(j.deleted).to.eql(true); });" - ] } } ] }, { - "name": "[PREVIEW] Vertex: mint resumable session (native)", - "request": { - "method": "POST", - "header": [], - "body": { - "mode": "formdata", - "formdata": [ - { - "key": "purpose", - "value": "user_data", - "type": "text" - }, + "name": "Cross-Cut Round 30: Opus 4.8 Feature Gating", + "description": "Exercises the Anthropic request-surface features on Claude Opus 4.8, the live successor after Fable 5 / Mythos were discontinued (https://www.anthropic.com/news/fable-mythos-access).\nSurface covered: adaptive thinking, budget_tokens to adaptive, sampling-param handling (temperature/top_p/top_k), effort high/xhigh/max, structured outputs, task budgets, computer-use new-gen tools, web_search dynamic filtering, mid-conversation system messages, and fast mode.\nScripts are guarded on code<400 so the suite tolerates accounts without access to a given feature (same convention as the [PREVIEW] items).", + "item": [ + { + "name": "Native Opus 4.8: adaptive thinking", + "event": [ { - "key": "filename", - "value": "harness-video.bin", - "type": "text" + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": { \"type\": \"adaptive\" },\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Native Opus 4.8: output_config effort high", + "event": [ { - "key": "content_type", - "value": "application/octet-stream", - "type": "text" + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 1024,\n \"output_config\": {\"effort\": \"high\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve x^2 - 5x + 6 = 0\"}]\n}" }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Native Opus 4.8: output_config effort xhigh", + "event": [ { - "key": "gcs_bucket", - "value": "{{vertexGcsBucket}}", - "type": "text" + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 1024,\n \"output_config\": {\"effort\": \"xhigh\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve x^2 - 5x + 6 = 0\"}]\n}" }, - { - "key": "gcs_prefix", - "value": "{{vertexGcsPrefix}}", - "type": "text" + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] } - ] + } }, - "url": { - "raw": "{{baseUrl}}/v1/files?provider=vertex", - "host": [ - "{{baseUrl}}" - ], - "path": [ - "v1", - "files" - ], - "query": [ + { + "name": "Native Opus 4.8: output_config effort max", + "event": [ { - "key": "provider", - "value": "vertex" + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 1024,\n \"output_config\": {\"effort\": \"max\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve x^2 - 5x + 6 = 0\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] } - ] - } - }, - "event": [ - { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "var j = pm.response.json();", - "pm.test('Vertex resumable: status pending_upload', function () { pm.expect(j.status).to.eql('pending_upload'); });", - "pm.test('Vertex resumable: has upload_url', function () { pm.expect(j.upload_url).to.be.a('string').and.not.empty; });", - "pm.collectionVariables.set('vertexUploadUrl', j.upload_url);", - "pm.collectionVariables.set('vertexResumableId', encodeURIComponent(j.id));" - ] - } - } - ] - }, - { - "name": "[PREVIEW] Vertex: PUT bytes to GCS session (resumable, direct to GCS)", - "request": { - "method": "PUT", - "header": [], - "body": { - "mode": "file", - "file": { - "src": "tests/e2e/api/fixtures/sample.txt" } }, - "url": { - "raw": "{{vertexUploadUrl}}", - "host": [ - "{{vertexUploadUrl}}" - ] - } - }, - "event": [ { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "pm.test('Vertex resumable: GCS accepted bytes (2xx)', function () { pm.expect(pm.response.code).to.be.within(200, 299); });" - ] - } - } - ] - }, - { - "name": "[PREVIEW] Vertex: delete resumable file (native, cleanup)", - "request": { - "method": "DELETE", - "header": [], - "url": { - "raw": "{{baseUrl}}/v1/files/{{vertexResumableId}}?provider=vertex", - "host": [ - "{{baseUrl}}" - ], - "path": [ - "v1", - "files", - "{{vertexResumableId}}" - ], - "query": [ + "name": "Native Opus 4.8: output_config format json_schema (structured outputs)", + "event": [ { - "key": "provider", - "value": "vertex" + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 800,\n \"output_config\": {\"format\": {\"type\":\"json_schema\",\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}},\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a city\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] } - ] - } - }, - "event": [ - { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "var j = pm.response.json();", - "pm.test('Vertex resumable cleanup: deleted', function () { pm.expect(j.deleted).to.eql(true); });" - ] } - } - ] - } - ] - }, - { - "name": "11c. Vertex Batches (/openai + native passthrough)", - "description": "Vertex batch prediction is GCS-backed. CRUD (create/list/retrieve/cancel) runs via the OpenAI drop-in (/openai/v1/batches, routed by provider=vertex); the base64 batch ids round-trip through every endpoint. The last two rows exercise the native genai surface raw passthrough: a verbatim Vertex BatchPredictionJob body POSTed to .../batchPredictionJobs is forwarded as-is and the native job resource is returned. All [PREVIEW]-tagged: needs a gateway Vertex provider key + a GCS bucket, plus vertexProject/vertexLocation for the native rows. Set vertexGcsBucket, vertexProject and run with --env-var include_preview=1.", - "item": [ - { - "name": "[PREVIEW] Vertex: upload batch input (GCS, /openai)", - "request": { - "method": "POST", - "header": [], - "body": { - "mode": "formdata", - "formdata": [ + }, + { + "name": "Native Opus 4.8: speed:fast stripped (fast mode unsupported, no beta header)", + "event": [ { - "key": "file", - "src": "tests/e2e/api/fixtures/sample.jsonl", - "type": "file" + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 800,\n \"speed\": \"fast\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}" }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Native Opus 4.8: task_budget (beta task-budgets-2026-03-13)", + "event": [ { - "key": "purpose", - "value": "batch", - "type": "text" + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "task-budgets-2026-03-13" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"output_config\": {\"task_budget\": {\"type\":\"tokens\",\"total\":20000}},\n \"messages\": [{\"role\":\"user\",\"content\":\"Plan and solve a multi-step task.\"}]\n}" }, - { - "key": "provider", - "value": "vertex", - "type": "text" + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Native Opus 4.8: computer use new-gen tools (computer-use-2025-11-24)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "anthropic-beta", + "value": "computer-use-2025-11-24" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"tools\": [\n { \"type\": \"computer_20251124\", \"name\": \"computer\", \"display_width_px\": 1024, \"display_height_px\": 768, \"display_number\": 1 },\n { \"type\": \"bash_20250124\", \"name\": \"bash\" },\n { \"type\": \"text_editor_20250728\", \"name\": \"str_replace_based_edit_tool\" }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot of the desktop.\"}]\n}" }, - { - "key": "storage_config[gcs][bucket]", - "value": "{{vertexGcsBucket}}", - "type": "text" + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Native Opus 4.8: web_search dynamic filtering (web_search_20260209)", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Find recent AI papers.\"}],\n \"tools\": [{\"type\":\"web_search_20260209\",\"name\":\"web_search\",\"max_uses\":3}]\n}" }, - { - "key": "storage_config[gcs][prefix]", - "value": "{{vertexGcsPrefix}}", - "type": "text" + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] } - ] + } }, - "url": { - "raw": "{{baseUrl}}/openai/v1/files", - "host": [ - "{{baseUrl}}" - ], - "path": [ - "openai", - "v1", - "files" - ] - } - }, - "event": [ { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "var j = pm.response.json();", - "pm.test('Vertex batch input: gcs backend', function () { pm.expect(j.storage_backend).to.eql('gcs'); });", - "pm.test('Vertex batch input: has id', function () { pm.expect(j.id).to.be.a('string').and.not.empty; });", - "pm.collectionVariables.set('vertexBatchInputFileId', j.id);", - "pm.collectionVariables.set('vertexBatchInputGcsUri', j.storage_uri);" - ] - } - } - ] - }, - { - "name": "[PREVIEW] Vertex: create batch (/openai)", - "request": { - "method": "POST", - "header": [ - { - "key": "Content-Type", - "value": "application/json" - } - ], - "body": { - "mode": "raw", - "raw": "{\n \"input_file_id\": \"{{vertexBatchInputFileId}}\",\n \"endpoint\": \"/v1/chat/completions\",\n \"completion_window\": \"24h\",\n \"provider\": \"vertex\",\n \"model\": \"gemini-2.5-flash\",\n \"output_folder\": {\n \"url\": \"gs://{{vertexGcsBucket}}/{{vertexGcsPrefix}}batch-output\"\n }\n}", - "options": { - "raw": { - "language": "json" + "name": "Native Opus 4.8: mid-conv system message (ends array)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 512,\n \"system\": [{\"type\":\"text\",\"text\":\"You are a helpful assistant.\"}],\n \"messages\": [\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Hello\"}]},\n {\"role\":\"assistant\",\"content\":[{\"type\":\"text\",\"text\":\"Hi!\"}]},\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"How are you?\"}]},\n {\"role\":\"system\",\"content\":[{\"type\":\"text\",\"text\":\"Respond only in one word.\"}]}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] } } }, - "url": { - "raw": "{{baseUrl}}/openai/v1/batches", - "host": [ - "{{baseUrl}}" - ], - "path": [ - "openai", - "v1", - "batches" - ] - } - }, - "event": [ { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "var j = pm.response.json();", - "pm.test('Vertex create batch: object batch', function () { pm.expect(j.object).to.eql('batch'); });", - "pm.test('Vertex create batch: has id', function () { pm.expect(j.id).to.be.a('string').and.not.empty; });", - "pm.test('Vertex create batch: input_file_id round-trips', function () { pm.expect(j.input_file_id).to.eql(pm.collectionVariables.get('vertexBatchInputFileId')); });", - "pm.collectionVariables.set('vertexBatchId', j.id);" - ] - } - } - ] - }, - { - "name": "[PREVIEW] Vertex: list batches (/openai)", - "request": { - "method": "GET", - "header": [], - "url": { - "raw": "{{baseUrl}}/openai/v1/batches?provider=vertex", - "host": [ - "{{baseUrl}}" - ], - "path": [ - "openai", - "v1", - "batches" - ], - "query": [ + "name": "Native Opus 4.8: mid-conv system message (before assistant)", + "event": [ { - "key": "provider", - "value": "vertex" + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }" + ] + } } - ] - } - }, - "event": [ - { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "var j = pm.response.json();", - "pm.test('Vertex list batches: created batch present', function () { pm.expect((j.data || []).map(function (b) { return b.id; })).to.include(pm.collectionVariables.get('vertexBatchId')); });" - ] - } - } - ] - }, - { - "name": "[PREVIEW] Vertex: retrieve batch (/openai)", - "request": { - "method": "GET", - "header": [], - "url": { - "raw": "{{baseUrl}}/openai/v1/batches/{{vertexBatchId}}?provider=vertex", - "host": [ - "{{baseUrl}}" - ], - "path": [ - "openai", - "v1", - "batches", - "{{vertexBatchId}}" ], - "query": [ + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 512,\n \"system\": [{\"type\":\"text\",\"text\":\"You are a helpful assistant.\"}],\n \"messages\": [\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Hello\"}]},\n {\"role\":\"system\",\"content\":[{\"type\":\"text\",\"text\":\"From now on be very concise.\"}]},\n {\"role\":\"assistant\",\"content\":[{\"type\":\"text\",\"text\":\"Hi!\"}]},\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"How are you?\"}]}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] + } + } + }, + { + "name": "Native Opus 4.8: adaptive thinking (family parity)", + "event": [ { - "key": "provider", - "value": "vertex" + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-api-key", + "value": "{{anthropicKey}}" + }, + { + "key": "anthropic-version", + "value": "2023-06-01" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": { \"type\": \"adaptive\" },\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/anthropic/v1/messages", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "anthropic", + "v1", + "messages" + ] } - ] - } - }, - "event": [ - { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "var j = pm.response.json();", - "pm.test('Vertex retrieve batch: object batch', function () { pm.expect(j.object).to.eql('batch'); });", - "pm.test('Vertex retrieve batch: id matches create', function () { pm.expect(j.id).to.eql(pm.collectionVariables.get('vertexBatchId')); });" - ] - } - } - ] - }, - { - "name": "[PREVIEW] Vertex: cancel batch (/openai)", - "request": { - "method": "POST", - "header": [ - { - "key": "Content-Type", - "value": "application/json" } - ], - "body": { - "mode": "raw", - "raw": "{\n \"provider\": \"vertex\"\n}", - "options": { - "raw": { - "language": "json" + }, + { + "name": "Cross-cut: anthropic/claude-opus-4-8 adaptive thinking", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] } } }, - "url": { - "raw": "{{baseUrl}}/openai/v1/batches/{{vertexBatchId}}/cancel", - "host": [ - "{{baseUrl}}" - ], - "path": [ - "openai", - "v1", - "batches", - "{{vertexBatchId}}", - "cancel" - ] - } - }, - "event": [ { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "var j = pm.response.json();", - "pm.test('Vertex cancel batch: has id', function () { pm.expect(j.id).to.be.a('string').and.not.empty; });", - "pm.test('Vertex cancel batch: cancelling/cancelled', function () { pm.expect(['cancelling','cancelled']).to.include(j.status); });" - ] - } - } - ] - }, - { - "name": "[PREVIEW] Vertex: create batch RAW passthrough (native genai)", - "request": { - "method": "POST", - "header": [ - { - "key": "Content-Type", - "value": "application/json" - } - ], - "body": { - "mode": "raw", - "raw": "{\n \"displayName\": \"bifrost-harness-passthrough\",\n \"model\": \"publishers/google/models/gemini-2.5-flash\",\n \"inputConfig\": {\n \"instancesFormat\": \"jsonl\",\n \"gcsSource\": {\n \"uris\": [\n \"{{vertexBatchInputGcsUri}}\"\n ]\n }\n },\n \"outputConfig\": {\n \"predictionsFormat\": \"jsonl\",\n \"gcsDestination\": {\n \"outputUriPrefix\": \"gs://{{vertexGcsBucket}}/{{vertexGcsPrefix}}batch-output\"\n }\n }\n}", - "options": { - "raw": { - "language": "json" + "name": "Cross-cut: anthropic/claude-opus-4-8 enabled thinking → adaptive (budget_tokens removed)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] } } }, - "url": { - "raw": "{{baseUrl}}/genai/v1/projects/{{vertexProject}}/locations/{{vertexLocation}}/batchPredictionJobs", - "host": [ - "{{baseUrl}}" - ], - "path": [ - "genai", - "v1", - "projects", - "{{vertexProject}}", - "locations", - "{{vertexLocation}}", - "batchPredictionJobs" - ] - } - }, - "event": [ { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "// Sends a verbatim Vertex BatchPredictionJob body; Bifrost passes it through (UseRawRequestBody)", - "// and returns the native Vertex job resource. Proves the raw-passthrough path end to end.", - "var j = pm.response.json();", - "pm.test('Vertex RAW passthrough: name is a batchPredictionJobs resource', function () { pm.expect(j.name).to.be.a('string'); pm.expect(j.name).to.include('batchPredictionJobs'); });", - "pm.test('Vertex RAW passthrough: has JOB_STATE_ state', function () { pm.expect(j.state || '').to.match(/^JOB_STATE_/); });", - "var parts = (j.name || '').split('/'); pm.collectionVariables.set('vertexRawBatchId', parts[parts.length - 1]);" - ] - } - } - ] - }, - { - "name": "[PREVIEW] Vertex: delete batch RAW passthrough (native genai, cleanup)", - "request": { - "method": "DELETE", - "header": [], - "url": { - "raw": "{{baseUrl}}/genai/v1/projects/{{vertexProject}}/locations/{{vertexLocation}}/batchPredictionJobs/{{vertexRawBatchId}}", - "host": [ - "{{baseUrl}}" + "name": "Cross-cut: anthropic/claude-opus-4-8 disabled thinking → omitted (no 400)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } ], - "path": [ - "genai", - "v1", - "projects", - "{{vertexProject}}", - "locations", - "{{vertexLocation}}", - "batchPredictionJobs", - "{{vertexRawBatchId}}" - ] - } - }, - "event": [ - { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "pm.test('Vertex RAW passthrough cleanup: 2xx', function () { pm.expect(pm.response.code).to.be.within(200, 299); });" - ] + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 1024,\n \"thinking\": {\"type\":\"disabled\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } } - } - ] - }, - { - "name": "[PREVIEW] Vertex: delete batch input file (/openai, cleanup)", - "request": { - "method": "DELETE", - "header": [], - "url": { - "raw": "{{baseUrl}}/openai/v1/files/{{vertexBatchInputFileId}}?provider=vertex", - "host": [ - "{{baseUrl}}" - ], - "path": [ - "openai", - "v1", - "files", - "{{vertexBatchInputFileId}}" - ], - "query": [ + }, + { + "name": "Cross-cut: anthropic/claude-opus-4-8 sampling params stripped (temperature/top_p/top_k)", + "event": [ { - "key": "provider", - "value": "vertex" + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 1024,\n \"temperature\": 0.7,\n \"top_p\": 0.9,\n \"top_k\": 5,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] } - ] - } - }, - "event": [ - { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "var j = pm.response.json();", - "pm.test('Vertex batch input cleanup: deleted', function () { pm.expect(j.deleted).to.eql(true); });" - ] } - } - ] - } - ] - }, - { - "name": "12. Backlog Coverage (auto-added missing cases)", - "description": "Comprehensive coverage of features sourced from each provider's docs. Organized by provider. Many entries will fail in environments without the corresponding model/feature provisioned - that's expected; they exist to surface gaps via the failure report.", - "item": [ - { - "name": "OpenAI Backlog", - "item": [ - { "name": "OpenAI: tool_choice specific function", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a color\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"pick_color\",\"parameters\":{\"type\":\"object\",\"properties\":{\"hex\":{\"type\":\"string\"}}}}}],\n \"tool_choice\": {\"type\":\"function\",\"function\":{\"name\":\"pick_color\"}}\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]}}}, - { "name": "OpenAI: parallel_tool_calls=false", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"What's the weather in NYC and SF?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}}],\n \"parallel_tool_calls\": false\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]}}}, - { "name": "OpenAI: response_format json_object", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"system\",\"content\":\"Output JSON only.\"},{\"role\":\"user\",\"content\":\"Tokyo population\"}],\n \"response_format\": {\"type\":\"json_object\"}\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]}}}, - { "name": "OpenAI: logprobs + top_logprobs", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"logprobs\": true,\n \"top_logprobs\": 5\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]}}}, - { "name": "OpenAI: seed for deterministic output", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a number\"}],\n \"seed\": 12345\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]}}}, - { "name": "OpenAI: stop sequences", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four, five\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]}}}, - { "name": "OpenAI: stream_options include_usage", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"stream\": true,\n \"stream_options\": {\"include_usage\": true}\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]}}}, - { "name": "OpenAI: predicted outputs", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Echo: hello world\"}],\n \"prediction\": {\"type\":\"content\",\"content\":\"hello world\"}\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]}}}, - { "name": "OpenAI: service_tier auto", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"service_tier\": \"auto\"\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]}}}, - { "name": "OpenAI: store + metadata", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"store\": true,\n \"metadata\": {\"harness\": \"backlog\"}\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]}}}, - { "name": "OpenAI Responses: reasoning summary", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"o3-mini\",\n \"input\": \"What's 17*23?\",\n \"reasoning\": {\"summary\": \"auto\"}\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/responses","host":["{{baseUrl}}"],"path":["openai","v1","responses"]}}}, - { "name": "OpenAI Responses streaming: summary_index + obfuscation preserved", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code >= 400) { return; }","pm.test('summary_index and obfuscation survive stream', function () {"," var body = pm.response.text() || '';"," pm.expect(body, 'expected obfuscation in SSE body').to.include('\"obfuscation\"');"," // reasoning.summary is model-discretionary: o3-mini may emit no summary on simple prompts."," // Only assert summary_index preservation when a reasoning summary was actually emitted."," if (body.indexOf('response.reasoning_summary') !== -1) {"," pm.expect(body, 'reasoning summary emitted but summary_index stripped').to.include('\"summary_index\"');"," }","});"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"o3-mini\",\n \"input\": \"Prove that the square root of 2 is irrational, step by step.\",\n \"reasoning\": {\"summary\": \"detailed\"},\n \"stream\": true,\n \"stream_options\": {\"include_obfuscation\": true}\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/responses","host":["{{baseUrl}}"],"path":["openai","v1","responses"]}}}, - { "name": "OpenAI Responses streaming: assistant phase preserved (gpt-5.3-codex)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code >= 400) { return; }","pm.test('phase appears on assistant message items', function () {"," var body = pm.response.text() || '';"," var hasPhase = body.indexOf('\"phase\":\"final_answer\"') !== -1 || body.indexOf('\"phase\":\"commentary\"') !== -1;"," pm.expect(hasPhase, 'no phase field in SSE body. First 200 chars: ' + body.slice(0,200)).to.be.true;","});"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-5.3-codex\",\n \"input\": \"Solve 2+2 and explain your steps briefly.\",\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/responses","host":["{{baseUrl}}"],"path":["openai","v1","responses"]}}}, - { "name": "OpenAI Responses: assistant phase input round-trip (gpt-5.3-codex)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code >= 400) {"," pm.test('phase field accepted on input message', function () {"," var body = (pm.response.text() || '').toLowerCase();"," pm.expect(body, 'unexpected rejection of phase field: ' + body.slice(0,200)).to.not.include('unknown field \"phase\"');"," pm.expect(body).to.not.include('unexpected field \"phase\"');"," });"," return;","}","pm.test('response returned output items', function () {"," var body = pm.response.text() || '';"," pm.expect(body).to.include('\"output\"');","});"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-5.3-codex\",\n \"input\": [\n {\"role\":\"user\",\"content\":\"What's 2+2?\"},\n {\"role\":\"assistant\",\"phase\":\"final_answer\",\"content\":\"4\"},\n {\"role\":\"user\",\"content\":\"Now what's 3+3?\"}\n ]\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/responses","host":["{{baseUrl}}"],"path":["openai","v1","responses"]}}}, - { "name": "OpenAI Responses: background mode", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"input\": \"Hi\",\n \"background\": true\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/responses","host":["{{baseUrl}}"],"path":["openai","v1","responses"]}}}, - { "name": "OpenAI Responses: truncation auto", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"input\": \"Hi\",\n \"truncation\": \"auto\"\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/responses","host":["{{baseUrl}}"],"path":["openai","v1","responses"]}}}, - { "name": "OpenAI Responses: include array", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"input\": \"Hi\",\n \"include\": [\"message.input_image.image_url\"]\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/responses","host":["{{baseUrl}}"],"path":["openai","v1","responses"]}}}, - { "name": "OpenAI Responses: custom tool", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"input\": \"Send Slack message\",\n \"tools\": [{\"type\":\"function\",\"name\":\"send_slack\",\"parameters\":{\"type\":\"object\",\"properties\":{\"channel\":{\"type\":\"string\"},\"message\":{\"type\":\"string\"}}}}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/responses","host":["{{baseUrl}}"],"path":["openai","v1","responses"]}}}, - { "name": "OpenAI Responses: token counting", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"input\": \"Count tokens for me\"\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/responses/input_tokens","host":["{{baseUrl}}"],"path":["openai","v1","responses","input_tokens"]}}} - ] - }, - { - "name": "Anthropic Backlog", - "item": [ - { "name": "Anthropic: prompt caching 1h TTL", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"extended-cache-ttl-2025-04-11"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"long context\",\"cache_control\":{\"type\":\"ephemeral\",\"ttl\":\"1h\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: web_fetch tool", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Fetch https://example.com and summarize\"}],\n \"tools\": [{\"type\":\"web_fetch_20250910\",\"name\":\"web_fetch\",\"max_uses\":2}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: memory tool", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Remember my name is Akshay\"}],\n \"tools\": [{\"type\":\"memory_20250818\",\"name\":\"memory\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: tool_search BM25", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"List tools matching 'weather'\"}],\n \"tools\": [{\"type\":\"tool_search_tool_bm25\",\"name\":\"tool_search\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: tool_search regex", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Find tools matching get_.*\"}],\n \"tools\": [{\"type\":\"tool_search_tool_regex\",\"name\":\"tool_search\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: code_execution v2 (20250825)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50! using Python\"}],\n \"tools\": [{\"type\":\"code_execution_20250825\",\"name\":\"code_execution\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: code_execution programmatic (20260120)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Plot sin(x) and tell me the period\"}],\n \"tools\": [{\"type\":\"code_execution_20260120\",\"name\":\"code_execution\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: PDF input", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: stop sequences", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop_sequences\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: service_tier auto", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"service_tier\": \"auto\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: output_config effort high", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-6\",\n \"max_tokens\": 1024,\n \"output_config\": {\"effort\": \"high\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve x^2 - 5x + 6 = 0\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: output_config format json_schema", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"output_config\": {\"format\": {\"type\":\"json_schema\",\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}},\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a city\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: tool defer_loading + advanced beta", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"advanced-tool-use-2025-09-15"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"tools\": [{\"name\":\"slow_tool\",\"input_schema\":{\"type\":\"object\"},\"defer_loading\":true},{\"name\":\"fast_tool\",\"input_schema\":{\"type\":\"object\"},\"defer_loading\":false}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Call slow_tool\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: tool input_examples + beta", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"tool-examples-2025-10-29"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 1024,\n \"tools\": [{\"name\":\"f\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"x\":{\"type\":\"number\"}}},\"input_examples\":[{\"input\":{\"x\":1},\"description\":\"basic\"}]}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Call f\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: strict tool input", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"tools\": [{\"name\":\"strict_fn\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"a\":{\"type\":\"string\"}},\"required\":[\"a\"],\"additionalProperties\":false},\"strict\":true}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Call strict_fn\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: token counting", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"How many tokens?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages/count_tokens","host":["{{baseUrl}}"],"path":["anthropic","v1","messages","count_tokens"]}}}, - { "name": "Anthropic: list models", "request": { "method": "GET", "header": [{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "url": {"raw":"{{baseUrl}}/anthropic/v1/models","host":["{{baseUrl}}"],"path":["anthropic","v1","models"]}}}, - { "name": "Anthropic: list batches", "request": { "method": "GET", "header": [{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "url": {"raw":"{{baseUrl}}/anthropic/v1/messages/batches","host":["{{baseUrl}}"],"path":["anthropic","v1","messages","batches"]}}}, - { "name": "Anthropic: list files", "request": { "method": "GET", "header": [{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"files-api-2025-04-14"}], "url": {"raw":"{{baseUrl}}/anthropic/v1/files","host":["{{baseUrl}}"],"path":["anthropic","v1","files"]}}} - ] - }, - { - "name": "Anthropic Beta Headers", - "item": [ - { "name": "Beta: token-efficient-tools", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"token-efficient-tools-2025-02-19"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"tools\": [{\"name\":\"f\",\"input_schema\":{\"type\":\"object\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Beta: fine-grained-tool-streaming", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"fine-grained-tool-streaming-2025-05-14"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"stream\": true,\n \"tools\": [{\"name\":\"f\",\"input_schema\":{\"type\":\"object\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "[PREVIEW] Beta: fast-mode (Opus 4.6)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"fast-mode-2026-02-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-6\",\n \"max_tokens\": 800,\n \"speed\": \"fast\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "[PREVIEW] Beta: fast-mode (Opus 4.7)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"fast-mode-2026-02-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 800,\n \"speed\": \"fast\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "[PREVIEW] Beta: fast-mode (Opus 4.8)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"fast-mode-2026-02-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 800,\n \"speed\": \"fast\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Beta: context-1m", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"context-1m-2025-09-15"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-6\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Beta: interleaved-thinking", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"interleaved-thinking-2025-05-14"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Beta: skills bundle", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"skills-2025-10-29"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Use skills\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Beta: redact-thinking", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"redact-thinking-2025-09-15"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Reasoned answer\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Beta: compaction", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"compact-2025-09-15"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-6\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Long convo\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Mid-conv system message (Opus 4.8) — system ends array", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 512,\n \"system\": [{\"type\":\"text\",\"text\":\"You are a helpful assistant.\"}],\n \"messages\": [\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Hello\"}]},\n {\"role\":\"assistant\",\"content\":[{\"type\":\"text\",\"text\":\"Hi!\"}]},\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"How are you?\"}]},\n {\"role\":\"system\",\"content\":[{\"type\":\"text\",\"text\":\"Respond only in one word.\"}]}\n ]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Mid-conv system message (Opus 4.8) — system mid-history (before assistant)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 512,\n \"system\": [{\"type\":\"text\",\"text\":\"You are a helpful assistant.\"}],\n \"messages\": [\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Hello\"}]},\n {\"role\":\"system\",\"content\":[{\"type\":\"text\",\"text\":\"From now on be very concise.\"}]},\n {\"role\":\"assistant\",\"content\":[{\"type\":\"text\",\"text\":\"Hi!\"}]},\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"How are you?\"}]}\n ]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}} - ] - }, - { - "name": "Bedrock Backlog", - "item": [ - { "name": "Bedrock Converse: streaming", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Count to 5\"}]}],\n \"inferenceConfig\": {\"maxTokens\": 256}\n}"}, "url": {"raw":"{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse-stream","host":["{{baseUrl}}"],"path":["bedrock","model","{{bedrockModel}}","converse-stream"]}}}, - { "name": "Bedrock Converse: stop sequences", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Count: one, two, three, four in lowercase\"}]}],\n \"inferenceConfig\": {\"maxTokens\": 256, \"stopSequences\": [\"three\"]}\n}"}, "url": {"raw":"{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse","host":["{{baseUrl}}"],"path":["bedrock","model","{{bedrockModel}}","converse"]}}}, - { "name": "Bedrock Converse: tool choice forced", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Pick a number\"}]}],\n \"toolConfig\": {\"tools\":[{\"toolSpec\":{\"name\":\"pick\",\"inputSchema\":{\"json\":{\"type\":\"object\",\"properties\":{\"n\":{\"type\":\"number\"}}}}}}],\"toolChoice\":{\"tool\":{\"name\":\"pick\"}}},\n \"inferenceConfig\": {\"maxTokens\":256}\n}"}, "url": {"raw":"{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse","host":["{{baseUrl}}"],"path":["bedrock","model","{{bedrockModel}}","converse"]}}}, - { "name": "Bedrock Converse: performance config optimized", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hi\"}]}],\n \"performanceConfig\": {\"latency\":\"optimized\"},\n \"inferenceConfig\": {\"maxTokens\":256}\n}"}, "url": {"raw":"{{baseUrl}}/bedrock/model/us.amazon.nova-pro-v1:0/converse","host":["{{baseUrl}}"],"path":["bedrock","model","us.amazon.nova-pro-v1:0","converse"]}}}, - { "name": "Bedrock Converse: request metadata", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hi\"}]}],\n \"requestMetadata\": {\"session\":\"harness\"},\n \"inferenceConfig\": {\"maxTokens\":256}\n}"}, "url": {"raw":"{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse","host":["{{baseUrl}}"],"path":["bedrock","model","{{bedrockModel}}","converse"]}}}, - { "name": "Bedrock InvokeModel: direct Anthropic shape", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"anthropic_version\": \"bedrock-2023-05-31\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/bedrock/model/{{bedrockModel}}/invoke","host":["{{baseUrl}}"],"path":["bedrock","model","{{bedrockModel}}","invoke"]}}} - ] - }, - { - "name": "Gemini Backlog", - "item": [ - { "name": "Gemini: tool config functionCallingConfig ANY", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Call get_weather\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"get_weather\",\"parameters\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"}}}}]}],\n \"toolConfig\": {\"functionCallingConfig\":{\"mode\":\"ANY\",\"allowedFunctionNames\":[\"get_weather\"]}}\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]}}}, - { "name": "Gemini: stop sequences", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Count: one, two, three, four in lowercase\"}]}],\n \"generationConfig\": {\"stopSequences\":[\"three\"]}\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]}}}, - { "name": "Gemini: temperature + topP + topK", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Pick a number\"}]}],\n \"generationConfig\": {\"temperature\":0.7,\"topP\":0.9,\"topK\":40}\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]}}}, - { "name": "Gemini: response logprobs", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Hi\"}]}],\n \"generationConfig\": {\"responseLogprobs\":true,\"logprobs\":3}\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]}}}, - { "name": "Gemini: presence + frequency penalty", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Tell a story\"}]}],\n \"generationConfig\": {\"presencePenalty\":0.5,\"frequencyPenalty\":0.5}\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]}}}, - { "name": "[PREVIEW] Gemini: PDF input", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Summarize this PDF\"},{\"fileData\":{\"mimeType\":\"application/pdf\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/data/A17_FlightPlan.pdf\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]}}}, - { "name": "Gemini: YouTube URL input", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Summarize this video\"},{\"fileData\":{\"fileUri\":\"https://www.youtube.com/watch?v=jNQXAC9IVRw\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]}}}, - { "name": "Gemini: URL context tool", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Summarize https://anthropic.com/news\"}]}],\n \"tools\": [{\"urlContext\":{}}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]}}}, - { "name": "Gemini: count tokens", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"How many tokens?\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:countTokens","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:countTokens"]}}}, - { "name": "Gemini: list models", "request": { "method": "GET", "header": [{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "url": {"raw":"{{baseUrl}}/genai/v1beta/models","host":["{{baseUrl}}"],"path":["genai","v1beta","models"]}}} - ] - }, - { - "name": "Vertex Backlog", - "item": [ - { "name": "Vertex: anthropic_version in body", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"anthropic_version\": \"vertex-2023-10-16\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi via Vertex\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex: streaming Anthropic", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "[PREVIEW] Vertex Model Garden: Llama (publishers/meta/...)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/meta/llama-4-maverick-17b-128e-instruct-maas\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "[PREVIEW] Vertex Model Garden: Mistral", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/publishers/mistralai/models/mistral-large-2411\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Azure Backlog", - "item": [ - { "name": "Azure: tools (function calling)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":\"What's the weather?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["openai","openai","deployments","{{azureDeployment}}","chat","completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]}}}, - { "name": "Azure: streaming", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["openai","openai","deployments","{{azureDeployment}}","chat","completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]}}}, - { "name": "Azure: structured output json_schema", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country for Paris\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"}},\"required\":[\"city\",\"country\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["openai","openai","deployments","{{azureDeployment}}","chat","completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]}}}, - { "name": "Azure: vision (image_url)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["openai","openai","deployments","{{azureDeployment}}","chat","completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]}}}, - { "name": "Azure: system + multi-turn", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Arrr!\"},{\"role\":\"user\",\"content\":\"Tell a joke\"}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["openai","openai","deployments","{{azureDeployment}}","chat","completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]}}}, - { "name": "Azure On Your Data: azure_search", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":\"What's in the docs?\"}],\n \"data_sources\": [{\"type\":\"azure_search\",\"parameters\":{\"endpoint\":\"https://placeholder.search.windows.net\",\"index_name\":\"placeholder\",\"authentication\":{\"type\":\"api_key\",\"key\":\"placeholder\"}}}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["openai","openai","deployments","{{azureDeployment}}","chat","completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]}}} - ] - }, - { - "name": "Cross-Provider Backlog", - "item": [ - { "name": "Cross-cut: code execution Anthropic", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50!\"}],\n \"tools\": [{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: code execution Gemini", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50!\"}],\n \"tools\": [{\"type\":\"code_execution\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: extended thinking via cross-model", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Plan a trip\"}],\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"max_tokens\": 4096\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: prompt caching via cross-model", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"system\",\"content\":[{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}]},{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: stop sequences (OpenAI)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked into content').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter'], 'unexpected finish_reason: ' + fr).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: stop sequences (Anthropic)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked into content').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter'], 'unexpected finish_reason: ' + fr).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: stop sequences (Gemini)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked into content').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter'], 'unexpected finish_reason: ' + fr).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: tool_choice forced (OpenAI)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"f\",\"parameters\":{\"type\":\"object\"}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: tool_choice forced (Bedrock)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"f\",\"parameters\":{\"type\":\"object\"}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: structured output Anthropic", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a city\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: structured output Bedrock", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a city\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vision Bedrock", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vision Vertex (Gemini)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: web search Bedrock (Anthropic)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: failover scenario (rate-limit-trigger)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"X-Bifrost-Fallback-Models","value":"openai/gpt-4o-mini,anthropic/claude-haiku-4-5"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: virtual key auth (X-Bifrost-VK)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"X-Bifrost-VK","value":"vk-test"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: sampling-params silently dropped for Opus 4.7", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"temperature\": 0.7,\n \"top_p\": 0.9\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: sampling-params silently dropped for Opus 4.8", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"temperature\": 0.7,\n \"top_p\": 0.9\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Passthrough Backlog", - "item": [ - { "name": "Passthrough OpenAI: web_search", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"input\": \"Latest AI news\",\n \"tools\": [{\"type\":\"web_search_preview\"}]\n}"}, "url": {"raw":"{{baseUrl}}/openai_passthrough/v1/responses","host":["{{baseUrl}}"],"path":["openai_passthrough","v1","responses"]}}}, - { "name": "Passthrough OpenAI: code_interpreter", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"input\": \"Plot sin(x)\",\n \"tools\": [{\"type\":\"code_interpreter\",\"container\":{\"type\":\"auto\"}}]\n}"}, "url": {"raw":"{{baseUrl}}/openai_passthrough/v1/responses","host":["{{baseUrl}}"],"path":["openai_passthrough","v1","responses"]}}}, - { "name": "Passthrough Anthropic: computer_use", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"computer-use-2025-11-24"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"tools\": [{\"type\":\"computer_20251124\",\"name\":\"computer\",\"display_width_px\":1024,\"display_height_px\":768},{\"type\":\"text_editor_20250728\",\"name\":\"str_replace_based_edit_tool\"},{\"type\":\"bash_20250124\",\"name\":\"bash\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic_passthrough/v1/messages","host":["{{baseUrl}}"],"path":["anthropic_passthrough","v1","messages"]}}}, - { "name": "Passthrough Anthropic: extended thinking", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic_passthrough/v1/messages","host":["{{baseUrl}}"],"path":["anthropic_passthrough","v1","messages"]}}}, - { "name": "Passthrough Anthropic: prompt caching", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic_passthrough/v1/messages","host":["{{baseUrl}}"],"path":["anthropic_passthrough","v1","messages"]}}}, - { "name": "Passthrough Anthropic: web_search", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic_passthrough/v1/messages","host":["{{baseUrl}}"],"path":["anthropic_passthrough","v1","messages"]}}}, - { "name": "Passthrough GenAI: googleSearch", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Latest news\"}]}],\n \"tools\": [{\"googleSearch\":{}}]\n}"}, "url": {"raw":"{{baseUrl}}/genai_passthrough/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai_passthrough","v1beta","models","{{genaiModel}}:generateContent"]}}}, - { "name": "Passthrough GenAI: codeExecution", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Compute fib(20)\"}]}],\n \"tools\": [{\"codeExecution\":{}}]\n}"}, "url": {"raw":"{{baseUrl}}/genai_passthrough/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai_passthrough","v1beta","models","{{genaiModel}}:generateContent"]}}}, - { "name": "Passthrough Azure: tools", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":\"What's the weather?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}}]\n}"}, "url": {"raw":"{{baseUrl}}/azure_passthrough/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["azure_passthrough","openai","deployments","{{azureDeployment}}","chat","completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]}}}, - { "name": "Passthrough OpenAI: streaming chat", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/openai_passthrough/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai_passthrough","v1","chat","completions"]}}}, - { "name": "Passthrough OpenAI: vision", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/openai_passthrough/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai_passthrough","v1","chat","completions"]}}} - ] - }, - { - "name": "Vertex Backlog Round 2 (Gemini-on-Vertex variants for system/multi-turn/sampling/tools)", - "item": [ - { "name": "Vertex Gemini: system instruction", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"systemInstruction\": {\"parts\":[{\"text\":\"You are a chef\"}]},\n \"contents\": [{\"parts\":[{\"text\":\"I have eggs\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{vertexModel}}:generateContent"]}}}, - { "name": "Vertex Gemini: multi-turn history", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"role\":\"user\",\"parts\":[{\"text\":\"Hi\"}]},{\"role\":\"model\",\"parts\":[{\"text\":\"Hello\"}]},{\"role\":\"user\",\"parts\":[{\"text\":\"How are you?\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{vertexModel}}:generateContent"]}}}, - { "name": "Vertex Gemini: stop sequences", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Count: one, two, three, four in lowercase\"}]}],\n \"generationConfig\": {\"stopSequences\":[\"three\"]}\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{vertexModel}}:generateContent"]}}}, - { "name": "Vertex Gemini: sampling params", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Hi\"}]}],\n \"generationConfig\": {\"temperature\":0.7,\"topP\":0.9,\"topK\":40}\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{vertexModel}}:generateContent"]}}}, - { "name": "Vertex Gemini: tool choice forced (ANY)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Hi\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"f\",\"parameters\":{\"type\":\"OBJECT\"}}]}],\n \"toolConfig\": {\"functionCallingConfig\":{\"mode\":\"ANY\"}}\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{vertexModel}}:generateContent"]}}}, - { "name": "Vertex Gemini: streaming", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Count 1-5\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:streamGenerateContent?alt=sse","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{vertexModel}}:streamGenerateContent"],"query":[{"key":"alt","value":"sse"}]}}}, - { "name": "Vertex Gemini: vision", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{vertexModel}}:generateContent"]}}}, - { "name": "Vertex Gemini: code execution", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Compute fib(20)\"}]}],\n \"tools\": [{\"codeExecution\":{}}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{vertexModel}}:generateContent"]}}}, - { "name": "Vertex Gemini: thinking budget", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Solve 17*23\"}]}],\n \"generationConfig\": {\"thinkingConfig\":{\"thinkingBudget\":4000}}\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{vertexModel}}:generateContent"]}}}, - { "name": "Vertex Gemini: structured output json_schema (via /v1/chat)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a city\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: extended thinking", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: web search", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: prompt caching", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "[PREVIEW] Vertex Claude: computer use", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"tools\": [{\"type\":\"computer_20250124\",\"name\":\"computer\",\"display_width_px\":1024,\"display_height_px\":768},{\"type\":\"text_editor_20250728\",\"name\":\"str_replace_based_edit_tool\"},{\"type\":\"bash_20250124\",\"name\":\"bash\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Bedrock Backlog Round 2 (Anthropic features via Bedrock)", - "item": [ - { "name": "Bedrock: multi-turn", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hi\"}]},{\"role\":\"assistant\",\"content\":[{\"text\":\"Hello\"}]},{\"role\":\"user\",\"content\":[{\"text\":\"How are you?\"}]}],\n \"inferenceConfig\": {\"maxTokens\":256}\n}"}, "url": {"raw":"{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse","host":["{{baseUrl}}"],"path":["bedrock","model","{{bedrockModel}}","converse"]}}}, - { "name": "Bedrock Converse: vision (image content)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Describe\"},{\"image\":{\"format\":\"jpeg\",\"source\":{\"bytes\":\"REPLACE_WITH_BASE64\"}}}]}],\n \"inferenceConfig\": {\"maxTokens\":512}\n}"}, "url": {"raw":"{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse","host":["{{baseUrl}}"],"path":["bedrock","model","{{bedrockModel}}","converse"]}}}, - { "name": "Bedrock via /v1: extended thinking (Sonnet 4.6)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock via /v1: adaptive thinking (Opus 4.7)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock via /v1: adaptive thinking (Opus 4.8)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock via /v1: mid-conv system fallback (Opus 4.8 — merged to top-level system)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"},\n {\"role\":\"system\",\"content\":\"Respond only in one word.\"}\n ]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock via /v1: web search (Anthropic)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock via /v1: code execution (Anthropic)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50!\"}],\n \"tools\": [{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock via /v1: prompt caching (Anthropic)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock via /v1: computer use", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"tools\": [{\"type\":\"computer_20251124\",\"name\":\"computer\",\"display_width_px\":1024,\"display_height_px\":768},{\"type\":\"text_editor_20250728\",\"name\":\"str_replace_based_edit_tool\"},{\"type\":\"bash_20250124\",\"name\":\"bash\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Screenshot\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock via /v1: PDF input (Anthropic)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock Converse: sampling params", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hi\"}]}],\n \"inferenceConfig\": {\"maxTokens\":256, \"temperature\":0.7}\n}"}, "url": {"raw":"{{baseUrl}}/bedrock/model/global.anthropic.claude-sonnet-4-6/converse","host":["{{baseUrl}}"],"path":["bedrock","model","global.anthropic.claude-sonnet-4-6","converse"]}}} - ] - }, - { - "name": "OpenAI/Azure Backlog Round 2", - "item": [ - { "name": "OpenAI: sampling params", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"temperature\": 0.7,\n \"top_p\": 0.9,\n \"frequency_penalty\": 0.5,\n \"presence_penalty\": 0.3\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]}}}, - { "name": "[PREVIEW] OpenAI Responses: MCP tool", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"input\": \"Use mcp\",\n \"tools\": [{\"type\":\"mcp\",\"server_label\":\"test\",\"server_url\":\"https://example.com/mcp\"}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/responses","host":["{{baseUrl}}"],"path":["openai","v1","responses"]}}}, - { "name": "[PREVIEW] OpenAI Responses: computer_use_preview", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"computer-use-preview\",\n \"input\": \"Take a screenshot\",\n \"truncation\": \"auto\",\n \"tools\": [{\"type\":\"computer_use_preview\",\"display_width\":1024,\"display_height\":768,\"environment\":\"browser\"}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/responses","host":["{{baseUrl}}"],"path":["openai","v1","responses"]}}}, - { "name": "[PREVIEW] OpenAI Responses: PDF input via input_file", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"input\": [{\"role\":\"user\",\"content\":[{\"type\":\"input_text\",\"text\":\"Summarize\"},{\"type\":\"input_file\",\"file_id\":\"file_REPLACE_ME\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/responses","host":["{{baseUrl}}"],"path":["openai","v1","responses"]}}}, - { "name": "[PREVIEW] OpenAI: audio input (gpt-4o-audio)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-audio-preview\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe audio\"},{\"type\":\"input_audio\",\"input_audio\":{\"data\":\"REPLACE_BASE64\",\"format\":\"wav\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]}}}, - { "name": "OpenAI: prompt caching marker (system reuse)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a helpful assistant. Long instructions for caching benefit. Long instructions for caching benefit. Long instructions for caching benefit.\"},{\"role\":\"user\",\"content\":\"Hi\"}],\n \"prompt_cache_key\": \"harness-cache-1\"\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai","v1","chat","completions"]}}}, - { "name": "OpenAI: list batches", "request": { "method": "GET", "header": [{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "url": {"raw":"{{baseUrl}}/openai/v1/batches","host":["{{baseUrl}}"],"path":["openai","v1","batches"]}}}, - { "name": "OpenAI: list files", "request": { "method": "GET", "header": [{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "url": {"raw":"{{baseUrl}}/openai/v1/files","host":["{{baseUrl}}"],"path":["openai","v1","files"]}}}, - { "name": "OpenAI: list models", "request": { "method": "GET", "header": [{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "url": {"raw":"{{baseUrl}}/openai/v1/models","host":["{{baseUrl}}"],"path":["openai","v1","models"]}}}, - { "name": "Azure: stop sequences", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["openai","openai","deployments","{{azureDeployment}}","chat","completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]}}}, - { "name": "Azure: sampling params", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"temperature\": 0.7,\n \"top_p\": 0.9\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["openai","openai","deployments","{{azureDeployment}}","chat","completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]}}}, - { "name": "Azure: tool_choice forced", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"f\",\"parameters\":{\"type\":\"object\"}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["openai","openai","deployments","{{azureDeployment}}","chat","completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]}}}, - { "name": "Azure: parallel_tool_calls=false", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"f\",\"parameters\":{\"type\":\"object\"}}}],\n \"parallel_tool_calls\": false\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["openai","openai","deployments","{{azureDeployment}}","chat","completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]}}}, - { "name": "[PREVIEW] Azure: reasoning_effort (o3 deployment)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"reasoning_effort\": \"high\"\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/o3/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["openai","openai","deployments","o3","chat","completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]}}} - ] - }, - { - "name": "Anthropic + Gemini Backlog Round 2", - "item": [ - { "name": "Anthropic: explicit multi-turn (3 turns)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: parallel tool calls disabled", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in NYC and SF?\"}],\n \"tools\": [{\"name\":\"get_weather\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}],\n \"tool_choice\": {\"type\":\"auto\",\"disable_parallel_tool_use\":true}\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "[PREVIEW] Anthropic: MCP toolset", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"mcp-client-2025-09-15"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"mcp_servers\": [{\"type\":\"url\",\"url\":\"https://example.com/mcp\",\"name\":\"test\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Use MCP\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: citations on document", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"text\",\"media_type\":\"text/plain\",\"data\":\"The sky is blue. Grass is green.\"},\"citations\":{\"enabled\":true}},{\"type\":\"text\",\"text\":\"What color is the sky?\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: eager input streaming + beta", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"eager-input-streaming-2025-10-29"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"stream\": true,\n \"tools\": [{\"name\":\"f\",\"input_schema\":{\"type\":\"object\"},\"eager_input_streaming\":true}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: allowed_callers + advanced beta", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"advanced-tool-use-2025-09-15"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 800,\n \"tools\": [{\"name\":\"f\",\"input_schema\":{\"type\":\"object\"},\"allowed_callers\":[\"direct\"]}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Anthropic: skills/container", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"skills-2025-10-29"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"container\": {\"skills\":[{\"skill_id\":\"data-analysis\",\"type\":\"anthropic\"}]},\n \"messages\": [{\"role\":\"user\",\"content\":\"Analyze\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Gemini: parallel function calls", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Weather in NYC and SF?\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"get_weather\",\"parameters\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"}}}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]}}}, - { "name": "Gemini: structured output via /v1/chat (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a city\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "[PREVIEW] Gemini: cached content reference", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Use cache\"}]}],\n \"cachedContent\": \"cachedContents/REPLACE_ME\"\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]}}}, - { "name": "Gemini: audio input (inline)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Transcribe\"},{\"inlineData\":{\"mimeType\":\"audio/wav\",\"data\":\"UklGRkQDAABXQVZFZm10IBAAAAABAAEAQB8AAIA+AAACABAAZGF0YSADAAAAAJUK6xPlGrIe3R5iG6oUgAv7AFj22eye5YHhAOEo5Jzql/MK/rcIXhLYGUQeHB9GHBgWTg3wAjv4cO645v7h0OBS4znp0fEW/NEGvhCxGLgdPB8OHXAXDg/jBCX6GPDs55niwOCZ4uznGPAl+uMEDg9wFw4dPB+4HbEYvhDRBhb80fE56VLj0OD+4bjmcO47+PACTg0YFkYcHB9EHtgZXhK3CAr+l/Oc6ijkAOGB4Z7l2exY9vsAgAuqFGIb3R6yHuUa6xOVCgAAa/UV7BvlTuEj4Z7kVuuA9AX/qAknE2Iafx4AH9gbZBVpDPYBSfei7SjmvOHk4Lrj6Omy8hD9xQeQEUgZAh4wH64cxxYvDuoDL/lC70/nSOLE4PLikOjy8B372wXoDxQYZx1AH2cdFBjoD9sFHfvy8JDo8uLE4EjiT+dC7y/56gMvDscWrhwwHwIeSBmQEcUHEP2y8ujpuuPk4LzhKOai7Un39gFpDGQV2BsAH38eYhonE6gJBf+A9FbrnuQj4U7hG+UV7Gv1AACVCusT5RqyHt0eYhuqFIAL+wBY9tnsnuWB4QDhKOSc6pfzCv63CF4S2BlEHhwfRhwYFk4N8AI7+HDuuOb+4dDgUuM56dHxFvzRBr4QsRi4HTwfDh1wFw4P4wQl+hjw7OeZ4sDgmeLs5xjwJfrjBA4PcBcOHTwfuB2xGL4Q0QYW/NHxOelS49Dg/uG45nDuO/jwAk4NGBZGHBwfRB7YGV4StwgK/pfznOoo5ADhgeGe5dnsWPb7AIALqhRiG90esh7lGusTlQoAAGv1Fewb5U7hI+Ge5FbrgPQF/6gJJxNiGn8eAB/YG2QVaQz2AUn3ou0o5rzh5OC64+jpsvIQ/cUHkBFIGQIeMB+uHMcWLw7qAy/5Qu9P50jixODy4pDo8vAd+9sF6A8UGGcdQB9nHRQY6A/bBR378vCQ6PLixOBI4k/nQu8v+eoDLw7HFq4cMB8CHkgZkBHFBxD9svLo6brj5OC84Sjmou1J9/YBaQxkFdgbAB9/HmIaJxOoCQX/gPRW657kI+FO4RvlFexr9Q==\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]}}}, - { "name": "Gemini: list cached contents", "request": { "method": "GET", "header": [{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "url": {"raw":"{{baseUrl}}/genai/v1beta/cachedContents","host":["{{baseUrl}}"],"path":["genai","v1beta","cachedContents"]}}}, - { "name": "Gemini: list files", "request": { "method": "GET", "header": [{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "url": {"raw":"{{baseUrl}}/genai/v1beta/files","host":["{{baseUrl}}"],"path":["genai","v1beta","files"]}}} - ] - }, - { - "name": "Bedrock Round 3 (every remaining gap)", - "item": [ - { "name": "Bedrock: stop sequences via /v1 (Anthropic)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: parallel_tool_calls disabled", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"f\",\"parameters\":{\"type\":\"object\"}}}],\n \"parallel_tool_calls\": false\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: MCP toolset", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"mcp_servers\": [{\"type\":\"url\",\"url\":\"https://example.com/mcp\",\"name\":\"test\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Use MCP\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: web_search dynamic filtering", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"messages\": [{\"role\":\"user\",\"content\":\"Search\"}],\n \"tools\": [{\"type\":\"web_search_20260209\",\"name\":\"web_search\"},{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: web_search domain filter", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Search\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2,\"allowed_domains\":[\"arxiv.org\"]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: web_search user_location", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Local search\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"user_location\":{\"type\":\"approximate\",\"city\":\"SF\",\"country\":\"US\"}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: web_fetch tool", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Fetch URL\"}],\n \"tools\": [{\"type\":\"web_fetch_20250910\",\"name\":\"web_fetch\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: text_editor tool", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"tools\": [{\"type\":\"text_editor_20250728\",\"name\":\"str_replace_based_edit_tool\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Edit\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: bash tool", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"tools\": [{\"type\":\"bash_20250124\",\"name\":\"bash\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Run ls\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: memory tool", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"tools\": [{\"type\":\"memory_20250818\",\"name\":\"memory\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Remember\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: citations on document", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"text\",\"media_type\":\"text/plain\",\"data\":\"Sky is blue\"},\"citations\":{\"enabled\":true}},{\"type\":\"text\",\"text\":\"What color?\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: prompt caching 1h", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long\",\"cache_control\":{\"type\":\"ephemeral\",\"ttl\":\"1h\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: anthropic-beta header", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-beta","value":"prompt-caching-2024-07-31"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: interleaved thinking", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-beta","value":"interleaved-thinking-2025-05-14"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: context management", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-beta","value":"context-management-2025-09-15"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: service_tier auto", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"max_tokens\": 800,\n \"service_tier\": \"default\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Bedrock: list invocation jobs (Batch)", "request": { "method": "GET", "header": [], "url": {"raw":"{{baseUrl}}/bedrock/model-invocation-jobs","host":["{{baseUrl}}"],"path":["bedrock","model-invocation-jobs"]}}} - ] - }, - { - "name": "Vertex Round 3 (every remaining gap)", - "item": [ - { "name": "Vertex Gemini: parallel function calls", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Weather NYC and SF\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"get_weather\",\"parameters\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"}}}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{vertexModel}}:generateContent"]}}}, - { "name": "Vertex Claude: tool_search", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Find tools\"}],\n \"tools\": [{\"type\":\"tool_search_tool_bm25\",\"name\":\"tool_search_tool_bm25\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: MCP toolset", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"mcp_servers\": [{\"type\":\"url\",\"url\":\"https://example.com/mcp\",\"name\":\"t\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Use MCP\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: web search dynamic", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"messages\": [{\"role\":\"user\",\"content\":\"Search\"}],\n \"tools\": [{\"type\":\"web_search_20260209\",\"name\":\"web_search\"},{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: web search domain", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Search\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"allowed_domains\":[\"arxiv.org\"]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: web search location", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Local\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"user_location\":{\"type\":\"approximate\",\"city\":\"SF\",\"country\":\"US\"}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: web_fetch", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Fetch URL\"}],\n \"tools\": [{\"type\":\"web_fetch_20250910\",\"name\":\"web_fetch\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: memory tool", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"tools\": [{\"type\":\"memory_20250818\",\"name\":\"memory\"}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Remember\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: PDF input", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Gemini: audio input", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Transcribe\"},{\"inlineData\":{\"mimeType\":\"audio/wav\",\"data\":\"UklGRkQDAABXQVZFZm10IBAAAAABAAEAQB8AAIA+AAACABAAZGF0YSADAAAAAJUK6xPlGrIe3R5iG6oUgAv7AFj22eye5YHhAOEo5Jzql/MK/rcIXhLYGUQeHB9GHBgWTg3wAjv4cO645v7h0OBS4znp0fEW/NEGvhCxGLgdPB8OHXAXDg/jBCX6GPDs55niwOCZ4uznGPAl+uMEDg9wFw4dPB+4HbEYvhDRBhb80fE56VLj0OD+4bjmcO47+PACTg0YFkYcHB9EHtgZXhK3CAr+l/Oc6ijkAOGB4Z7l2exY9vsAgAuqFGIb3R6yHuUa6xOVCgAAa/UV7BvlTuEj4Z7kVuuA9AX/qAknE2Iafx4AH9gbZBVpDPYBSfei7SjmvOHk4Lrj6Omy8hD9xQeQEUgZAh4wH64cxxYvDuoDL/lC70/nSOLE4PLikOjy8B372wXoDxQYZx1AH2cdFBjoD9sFHfvy8JDo8uLE4EjiT+dC7y/56gMvDscWrhwwHwIeSBmQEcUHEP2y8ujpuuPk4LzhKOai7Un39gFpDGQV2BsAH38eYhonE6gJBf+A9FbrnuQj4U7hG+UV7Gv1AACVCusT5RqyHt0eYhuqFIAL+wBY9tnsnuWB4QDhKOSc6pfzCv63CF4S2BlEHhwfRhwYFk4N8AI7+HDuuOb+4dDgUuM56dHxFvzRBr4QsRi4HTwfDh1wFw4P4wQl+hjw7OeZ4sDgmeLs5xjwJfrjBA4PcBcOHTwfuB2xGL4Q0QYW/NHxOelS49Dg/uG45nDuO/jwAk4NGBZGHBwfRB7YGV4StwgK/pfznOoo5ADhgeGe5dnsWPb7AIALqhRiG90esh7lGusTlQoAAGv1Fewb5U7hI+Ge5FbrgPQF/6gJJxNiGn8eAB/YG2QVaQz2AUn3ou0o5rzh5OC64+jpsvIQ/cUHkBFIGQIeMB+uHMcWLw7qAy/5Qu9P50jixODy4pDo8vAd+9sF6A8UGGcdQB9nHRQY6A/bBR378vCQ6PLixOBI4k/nQu8v+eoDLw7HFq4cMB8CHkgZkBHFBxD9svLo6brj5OC84Sjmou1J9/YBaQxkFdgbAB9/HmIaJxOoCQX/gPRW657kI+FO4RvlFexr9Q==\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{vertexModel}}:generateContent"]}}}, - { "name": "Vertex Claude: citations", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"text\",\"media_type\":\"text/plain\",\"data\":\"Sky blue\"},\"citations\":{\"enabled\":true}},{\"type\":\"text\",\"text\":\"Color?\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: extended thinking budget_tokens", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: prompt caching 1h", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long\",\"cache_control\":{\"type\":\"ephemeral\",\"ttl\":\"1h\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: anthropic-beta header", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-beta","value":"prompt-caching-2024-07-31"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: defer_loading", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"tools\": [{\"name\":\"f\",\"input_schema\":{\"type\":\"object\"},\"defer_loading\":true}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: allowed_callers", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"tools\": [{\"name\":\"f\",\"input_schema\":{\"type\":\"object\"},\"allowed_callers\":[\"a\"]}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: interleaved thinking", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-beta","value":"interleaved-thinking-2025-05-14"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: context management", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-beta","value":"context-management-2025-09-15"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Gemini: safety settings BLOCK_NONE", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Tell a story\"}]}],\n \"safetySettings\": [{\"category\":\"HARM_CATEGORY_DANGEROUS_CONTENT\",\"threshold\":\"BLOCK_NONE\"}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{vertexModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{vertexModel}}:generateContent"]}}} - ] - }, - { - "name": "OpenAI/Anthropic/Gemini/Azure Round 3 (final gap closure)", - "item": [ - { "name": "[PREVIEW] OpenAI: file_search w/ placeholder vector store", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"input\": \"Search KB\",\n \"tools\": [{\"type\":\"file_search\",\"vector_store_ids\":[\"vs_REPLACE_ME\"]}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/responses","host":["{{baseUrl}}"],"path":["openai","v1","responses"]}}}, - { "name": "OpenAI: token counting endpoint", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"input\": \"Count my tokens\"\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/responses/input_tokens","host":["{{baseUrl}}"],"path":["openai","v1","responses","input_tokens"]}}}, - { "name": "OpenAI: create batch", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"input_file_id\": \"file_REPLACE_ME\",\n \"endpoint\": \"/v1/chat/completions\",\n \"completion_window\": \"24h\"\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/batches","host":["{{baseUrl}}"],"path":["openai","v1","batches"]}}}, - { "name": "Anthropic: token counting endpoint", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"How many tokens in this message?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages/count_tokens","host":["{{baseUrl}}"],"path":["anthropic","v1","messages","count_tokens"]}}}, - { "name": "Anthropic: create batch", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"requests\": [{\"custom_id\":\"req-1\",\"params\":{\"model\":\"claude-haiku-4-5\",\"max_tokens\":256,\"messages\":[{\"role\":\"user\",\"content\":\"Hi\"}]}}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages/batches","host":["{{baseUrl}}"],"path":["anthropic","v1","messages","batches"]}}}, - { "name": "Gemini: tool_choice forced via toolConfig", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Hi\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"f\",\"parameters\":{\"type\":\"OBJECT\"}}]}],\n \"toolConfig\": {\"functionCallingConfig\":{\"mode\":\"ANY\",\"allowedFunctionNames\":[\"f\"]}}\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]}}}, - { "name": "[PREVIEW] Gemini: prompt caching via cachedContents", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Use cache\"}]}],\n \"cachedContent\": \"cachedContents/PLACEHOLDER\"\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:generateContent","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:generateContent"]}}}, - { "name": "Gemini: token counting", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"How many tokens?\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/{{genaiModel}}:countTokens","host":["{{baseUrl}}"],"path":["genai","v1beta","models","{{genaiModel}}:countTokens"]}}}, - { "name": "Azure: code_interpreter via Responses preview", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"input\": \"Plot sin(x)\",\n \"tools\": [{\"type\":\"code_interpreter\",\"container\":{\"type\":\"auto\"}}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/responses?api-version=2025-04-01-preview","host":["{{baseUrl}}"],"path":["openai","openai","deployments","{{azureDeployment}}","responses"],"query":[{"key":"api-version","value":"2025-04-01-preview"}]}}}, - { "name": "[PREVIEW] Azure: file_search via Responses preview", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"input\": \"Search\",\n \"tools\": [{\"type\":\"file_search\",\"vector_store_ids\":[\"vs_REPLACE_ME\"]}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/responses?api-version=2025-04-01-preview","host":["{{baseUrl}}"],"path":["openai","openai","deployments","{{azureDeployment}}","responses"],"query":[{"key":"api-version","value":"2025-04-01-preview"}]}}}, - { "name": "[PREVIEW] Azure: audio (gpt-4o-audio deployment)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"input_audio\",\"input_audio\":{\"data\":\"REPLACE_BASE64\",\"format\":\"wav\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/gpt-4o-audio-preview/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["openai","openai","deployments","gpt-4o-audio-preview","chat","completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]}}}, - { "name": "[PREVIEW] Azure: skills/container", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"input\": \"Use skills\",\n \"tools\": [{\"type\":\"code_interpreter\",\"container\":{\"type\":\"auto\"}}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/responses?api-version=2025-04-01-preview","host":["{{baseUrl}}"],"path":["openai","openai","deployments","{{azureDeployment}}","responses"],"query":[{"key":"api-version","value":"2025-04-01-preview"}]}}}, - { "name": "Azure: service_tier scale", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"service_tier\": \"flex\"\n}"}, "url": {"raw":"{{baseUrl}}/openai/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["openai","openai","deployments","{{azureDeployment}}","chat","completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]}}} - ] - }, - { - "name": "Cross-Cut Round 4 (Vertex Claude + Vertex Gemini + Azure cross-cut)", - "item": [ - { "name": "Vertex Claude: structured output (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (CityInfo)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('name').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p.name.toLowerCase(), 'expected Paris').to.include('paris'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"system\",\"content\":\"Extract the city information from the user's message.\"},{\"role\":\"user\",\"content\":\"I visited Paris, France last summer.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"CityInfo\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"name\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"}},\"required\":[\"name\",\"country\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: function calling cross-cut", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: streaming", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: vision", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"What is in this image?\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: tool_choice forced", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"f\",\"parameters\":{\"type\":\"object\"}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: stop sequences", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: multi-turn cross-cut", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: system message cross-cut", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: web search via /v1/chat (sonnet)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: code execution via /v1/chat (opus)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50!\"}],\n \"tools\": [{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: PDF input via /v1/chat (sonnet)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: sampling-params dropped for Opus 4.7", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"temperature\": 0.7,\n \"top_p\": 0.9\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Claude: sampling-params dropped for Opus 4.8", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-8\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"temperature\": 0.7,\n \"top_p\": 0.9\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Gemini: function calling cross-cut", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city argument', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls in response').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('arguments not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Vertex Gemini: streaming cross-cut", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: web search Vertex Gemini (google_search)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"google_search\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: stop sequences (Bedrock)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked into content').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter'], 'unexpected finish_reason: ' + fr).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: stop sequences (Vertex Gemini)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked into content').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter'], 'unexpected finish_reason: ' + fr).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: Azure basic chat", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: Azure tools", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: Azure structured output (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country for Paris\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"}},\"required\":[\"city\",\"country\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: Azure streaming", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: Azure vision", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Cross-Cut Round 5: Structured Output Matrix (response_format json_schema via /v1/chat across providers/models)", - "item": [ - { "name": "Cross-cut: openai/gpt-5 (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-5-mini (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-5-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-4o (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-4.1 (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4.1\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/o3-mini (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/o3-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-7 (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-sonnet-4-6 (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/claude-opus-4-7 (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/claude-haiku-4-5 (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "[PREVIEW] Cross-cut: bedrock/nova-pro (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/us.amazon.nova-pro-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "[PREVIEW] Cross-cut: bedrock/nova-lite (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-opus-4-7 (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/gemini-2.5-flash (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: gemini/gemini-2.5-pro (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: schema-compliant JSON (city/country/pop)', function () { var j = pm.response.json(); var c = ''; if (j.choices && j.choices[0] && j.choices[0].message) { c = j.choices[0].message.content || ''; } if (!c && Array.isArray(j.content)) { var tb = j.content.find(function (b) { return b.type === 'text' && b.text; }); c = tb ? tb.text : ''; } pm.expect(c, 'content was empty').to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('content not JSON: ' + e.message + ' (got: ' + c.slice(0,120) + ')'); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Cross-Cut Round 6: Function Calling Matrix", - "item": [ - { "name": "Cross-cut: openai/gpt-5 function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-5-mini function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-5-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-4o function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-4.1 function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4.1\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/o3-mini function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/o3-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-7 function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-sonnet-4-6 function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-haiku-4-5 function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0 function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "[PREVIEW] Cross-cut: bedrock/us.amazon.nova-pro-v1:0 function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/us.amazon.nova-pro-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-opus-4-7 function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-sonnet-4-6 function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/gemini-2.5-pro function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/gemini-2.5-flash function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: gemini/gemini-2.5-pro function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: gemini/gemini-2.5-flash function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: azure/{{azureDeployment}} function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked with city', function () { var j = pm.response.json(); var tc = null; if (j.choices && j.choices[0] && j.choices[0].message && Array.isArray(j.choices[0].message.tool_calls) && j.choices[0].message.tool_calls.length) { tc = j.choices[0].message.tool_calls[0]; } pm.expect(tc, 'no tool_calls').to.not.be.null; pm.expect(tc.function && tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Cross-Cut Round 7: Streaming Matrix", - "item": [ - { "name": "Cross-cut: openai/gpt-5 streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-5-mini streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-5-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-4o streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-4.1 streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4.1\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/o3-mini streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/o3-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-7 streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-sonnet-4-6 streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-haiku-4-5 streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0 streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-opus-4-7 streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-sonnet-4-6 streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/gemini-2.5-pro streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/gemini-2.5-flash streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: gemini/gemini-2.5-pro streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: gemini/gemini-2.5-flash streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: azure/{{azureDeployment}} streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: response is SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected SSE, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Cross-Cut Round 8: Vision Matrix", - "item": [ - { "name": "Cross-cut: openai/gpt-5 vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-4o vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-4.1 vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4.1\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-7 vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-sonnet-4-6 vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-haiku-4-5 vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-opus-4-7 vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-sonnet-4-6 vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/gemini-2.5-pro vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/gemini-2.5-flash vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: gemini/gemini-2.5-pro vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: gemini/gemini-2.5-flash vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: azure/{{azureDeployment}} vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Cross-Cut Round 9: Tool Choice Forced Matrix", - "item": [ - { "name": "Cross-cut: openai/gpt-5 tool_choice forced", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-5-mini tool_choice forced", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-5-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-4o tool_choice forced", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-4.1 tool_choice forced", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4.1\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-7 tool_choice forced", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-sonnet-4-6 tool_choice forced", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-haiku-4-5 tool_choice forced", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 tool_choice forced", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 tool_choice forced", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-opus-4-7 tool_choice forced", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-sonnet-4-6 tool_choice forced", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: gemini/gemini-2.5-pro tool_choice forced", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: azure/{{azureDeployment}} tool_choice forced", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Tool choice forced: tool_calls present', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls; pm.expect(Array.isArray(tc) && tc.length > 0).to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}],\n \"tool_choice\": \"required\"\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Cross-Cut Round 10: Stop Sequences Matrix", - "item": [ - { "name": "Cross-cut: openai/gpt-4o-mini stop sequences", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-4o stop sequences", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-7 stop sequences", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-sonnet-4-6 stop sequences", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-haiku-4-5 stop sequences", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 stop sequences", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 stop sequences", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-opus-4-7 stop sequences", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-sonnet-4-6 stop sequences", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/gemini-2.5-flash stop sequences", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: gemini/gemini-2.5-flash stop sequences", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Stop sequence: halted before stop token', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; var fr = (j.choices && j.choices[0] && j.choices[0].finish_reason) || ''; pm.expect(c.toLowerCase(), 'stop token \"three\" leaked').to.not.include('three'); pm.expect(['stop','stop_sequence','length','content_filter']).to.include(fr); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count: one, two, three, four in lowercase\"}],\n \"stop\": [\"three\"]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Cross-Cut Round 11: Multi-turn Matrix", - "item": [ - { "name": "Cross-cut: openai/gpt-5 multi-turn", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-4o multi-turn", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-7 multi-turn", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-sonnet-4-6 multi-turn", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-haiku-4-5 multi-turn", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 multi-turn", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 multi-turn", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-opus-4-7 multi-turn", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-sonnet-4-6 multi-turn", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/gemini-2.5-pro multi-turn", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: gemini/gemini-2.5-pro multi-turn", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Cross-Cut Round 12: System Message Matrix", - "item": [ - { "name": "Cross-cut: openai/gpt-5 system message", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-5\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: openai/gpt-4o system message", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"openai/gpt-4o\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-7 system message", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-sonnet-4-6 system message", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-haiku-4-5 system message", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-opus-4-7 system message", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/global.anthropic.claude-sonnet-4-6 system message", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-opus-4-7 system message", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-sonnet-4-6 system message", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/gemini-2.5-pro system message", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: gemini/gemini-2.5-pro system message", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-pro\",\n \"messages\": [{\"role\":\"system\",\"content\":\"You are a pirate.\"},{\"role\":\"user\",\"content\":\"Greet me\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Cross-Cut Round 13: Web Search Matrix", - "item": [ - { "name": "Cross-cut: anthropic/claude-opus-4-7 web_search", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Web search: response non-empty', function () { var raw = JSON.stringify(pm.response.json()); pm.expect(raw.length).to.be.greaterThan(100); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-sonnet-4-6 web_search", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Web search: response non-empty', function () { var raw = JSON.stringify(pm.response.json()); pm.expect(raw.length).to.be.greaterThan(100); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/claude-opus-4-7 web_search", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Web search: response non-empty', function () { var raw = JSON.stringify(pm.response.json()); pm.expect(raw.length).to.be.greaterThan(100); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-opus-4-7 web_search", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Web search: response non-empty', function () { var raw = JSON.stringify(pm.response.json()); pm.expect(raw.length).to.be.greaterThan(100); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":2}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: gemini/gemini-2.5-flash google_search", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Web search: response non-empty', function () { var raw = JSON.stringify(pm.response.json()); pm.expect(raw.length).to.be.greaterThan(100); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"google_search\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/gemini-2.5-flash google_search", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Web search: response non-empty', function () { var raw = JSON.stringify(pm.response.json()); pm.expect(raw.length).to.be.greaterThan(100); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Latest news\"}],\n \"tools\": [{\"type\":\"google_search\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Cross-Cut Round 14: Code Execution Matrix", - "item": [ - { "name": "Cross-cut: anthropic/claude-opus-4-7 code_execution", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50!\"}],\n \"tools\": [{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/claude-opus-4-7 code_execution", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50!\"}],\n \"tools\": [{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-opus-4-7 code_execution", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Compute 50!\"}],\n \"tools\": [{\"type\":\"code_execution_20250522\",\"name\":\"code_execution\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Cross-Cut Round 15: Extended/Adaptive Thinking Matrix", - "item": [ - { "name": "Cross-cut: anthropic/claude-opus-4-7 adaptive thinking", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-8 adaptive thinking", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-sonnet-4-6 enabled thinking", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/claude-opus-4-7 adaptive thinking", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/claude-opus-4-8 adaptive thinking", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/claude-sonnet-4-6 enabled thinking", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-opus-4-7 adaptive thinking", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-opus-4-8 adaptive thinking", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-sonnet-4-6 enabled thinking", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Cross-Cut Round 16: Prompt Caching Matrix", - "item": [ - { "name": "Cross-cut: anthropic/claude-opus-4-7 prompt caching", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Prompt caching: usage present', function () { var j = pm.response.json(); pm.expect(j.usage || {}).to.be.an('object'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-haiku-4-5 prompt caching", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Prompt caching: usage present', function () { var j = pm.response.json(); pm.expect(j.usage || {}).to.be.an('object'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/claude-opus-4-7 prompt caching", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Prompt caching: usage present', function () { var j = pm.response.json(); pm.expect(j.usage || {}).to.be.an('object'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/claude-haiku-4-5 prompt caching", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Prompt caching: usage present', function () { var j = pm.response.json(); pm.expect(j.usage || {}).to.be.an('object'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-haiku-4-5-20251001-v1:0\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-opus-4-7 prompt caching", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Prompt caching: usage present', function () { var j = pm.response.json(); pm.expect(j.usage || {}).to.be.an('object'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-sonnet-4-6 prompt caching", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Prompt caching: usage present', function () { var j = pm.response.json(); pm.expect(j.usage || {}).to.be.an('object'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-sonnet-4-6\",\n \"max_tokens\": 800,\n \"system\": [{\"type\":\"text\",\"text\":\"Long ctx\",\"cache_control\":{\"type\":\"ephemeral\"}}],\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Cross-Cut Round 17: PDF Input Matrix", - "item": [ - { "name": "Cross-cut: anthropic/claude-opus-4-7 PDF input", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-sonnet-4-6 PDF input", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-sonnet-4-6\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/claude-opus-4-7 PDF input", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-7\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-opus-4-7 PDF input", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-7\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"document\",\"source\":{\"type\":\"url\",\"url\":\"https://www.berkshirehathaway.com/letters/2024ltr.pdf\"}},{\"type\":\"text\",\"text\":\"Summarize\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} - ] - }, - { - "name": "Cross-Cut Round 18: Cohere Drop-in Smoke", - "item": [ - { "name": "Cohere drop-in: basic chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Cohere shape: message.content non-empty', function () { var j = pm.response.json(); var c = j.message && j.message.content; pm.expect(Array.isArray(c) ? c.length > 0 : (typeof c === 'string' && c.length > 0), 'expected non-empty cohere message content').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"command-r-plus\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}"}, "url": {"raw":"{{baseUrl}}/cohere/v2/chat","host":["{{baseUrl}}"],"path":["cohere", "v2", "chat"]} } }, - { "name": "Cohere drop-in: streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var raw = JSON.stringify(j); pm.expect(raw.length, 'body too small').to.be.greaterThan(50); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"command-r-plus\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}]\n,\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/cohere/v2/chat","host":["{{baseUrl}}"],"path":["cohere", "v2", "chat"]} } }, - { "name": "Cohere drop-in: multi-turn", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Cohere shape: message.content non-empty', function () { var j = pm.response.json(); var c = j.message && j.message.content; pm.expect(Array.isArray(c) ? c.length > 0 : (typeof c === 'string' && c.length > 0), 'expected non-empty cohere message content').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"command-r-plus\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/cohere/v2/chat","host":["{{baseUrl}}"],"path":["cohere", "v2", "chat"]} } }, - { "name": "Cohere drop-in: tools", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Cohere shape: message.content non-empty', function () { var j = pm.response.json(); var c = j.message && j.message.content; pm.expect(Array.isArray(c) ? c.length > 0 : (typeof c === 'string' && c.length > 0), 'expected non-empty cohere message content').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"command-r-plus\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/cohere/v2/chat","host":["{{baseUrl}}"],"path":["cohere", "v2", "chat"]} } }, - { "name": "Cohere drop-in: list models", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var raw = JSON.stringify(j); pm.expect(raw.length, 'body too small').to.be.greaterThan(50); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":""}, "url": {"raw":"{{baseUrl}}/cohere/v1/models","host":["{{baseUrl}}"],"path":["cohere", "v1", "models"]} } } - ] - }, - { - "name": "Cross-Cut Round 19: LangChain Drop-in Smoke", - "item": [ - { "name": "LangChain drop-in: OpenAI shape chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('OpenAI shape: choices[0].message.content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && (j.choices[0].message.content || (j.choices[0].message.tool_calls && j.choices[0].message.tool_calls.length))) || ''; pm.expect(c, 'no content or tool_calls').to.be.ok; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}"}, "url": {"raw":"{{baseUrl}}/langchain/v1/chat/completions","host":["{{baseUrl}}"],"path":["langchain", "v1", "chat", "completions"]} } }, - { "name": "LangChain drop-in: Anthropic shape chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic shape: content array non-empty', function () { var j = pm.response.json(); pm.expect(Array.isArray(j.content) && j.content.length > 0, 'expected non-empty content array').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}"}, "url": {"raw":"{{baseUrl}}/langchain/v1/messages","host":["{{baseUrl}}"],"path":["langchain", "v1", "messages"]} } }, - { "name": "LangChain drop-in: Gemini shape chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Gemini shape: candidates[0].content.parts non-empty', function () { var j = pm.response.json(); var parts = j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts; pm.expect(Array.isArray(parts) && parts.length > 0, 'expected non-empty parts').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Hello\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/langchain/v1beta/models/gemini-2.5-flash:generateContent","host":["{{baseUrl}}"],"path":["langchain", "v1beta", "models", "gemini-2.5-flash:generateContent"]} } }, - { "name": "LangChain drop-in: Bedrock shape converse", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Bedrock Converse shape: output.message.content non-empty', function () { var j = pm.response.json(); var content = j.output && j.output.message && j.output.message.content; pm.expect(Array.isArray(content) && content.length > 0, 'expected non-empty content').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hello\"}]}],\n \"inferenceConfig\": {\"maxTokens\": 256}\n}"}, "url": {"raw":"{{baseUrl}}/langchain/model/global.anthropic.claude-haiku-4-5-20251001-v1:0/converse","host":["{{baseUrl}}"],"path":["langchain", "model", "global.anthropic.claude-haiku-4-5-20251001-v1:0", "converse"]} } }, - { "name": "LangChain drop-in: Cohere shape chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Cohere shape: message.content non-empty', function () { var j = pm.response.json(); var c = j.message && j.message.content; pm.expect(Array.isArray(c) ? c.length > 0 : (typeof c === 'string' && c.length > 0), 'expected non-empty cohere message content').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"command-r-plus\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}"}, "url": {"raw":"{{baseUrl}}/langchain/v2/chat","host":["{{baseUrl}}"],"path":["langchain", "v2", "chat"]} } } - ] - }, - { - "name": "Cross-Cut Round 20: LiteLLM Drop-in Smoke", - "item": [ - { "name": "LiteLLM drop-in: OpenAI shape chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('OpenAI shape: choices[0].message.content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && (j.choices[0].message.content || (j.choices[0].message.tool_calls && j.choices[0].message.tool_calls.length))) || ''; pm.expect(c, 'no content or tool_calls').to.be.ok; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}"}, "url": {"raw":"{{baseUrl}}/litellm/v1/chat/completions","host":["{{baseUrl}}"],"path":["litellm", "v1", "chat", "completions"]} } }, - { "name": "LiteLLM drop-in: Anthropic shape chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic shape: content array non-empty', function () { var j = pm.response.json(); pm.expect(Array.isArray(j.content) && j.content.length > 0, 'expected non-empty content array').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}"}, "url": {"raw":"{{baseUrl}}/litellm/v1/messages","host":["{{baseUrl}}"],"path":["litellm", "v1", "messages"]} } }, - { "name": "LiteLLM drop-in: Gemini shape chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Gemini shape: candidates[0].content.parts non-empty', function () { var j = pm.response.json(); var parts = j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts; pm.expect(Array.isArray(parts) && parts.length > 0, 'expected non-empty parts').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Hello\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/litellm/v1beta/models/gemini-2.5-flash:generateContent","host":["{{baseUrl}}"],"path":["litellm", "v1beta", "models", "gemini-2.5-flash:generateContent"]} } }, - { "name": "LiteLLM drop-in: Bedrock shape converse", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Bedrock Converse shape: output.message.content non-empty', function () { var j = pm.response.json(); var content = j.output && j.output.message && j.output.message.content; pm.expect(Array.isArray(content) && content.length > 0, 'expected non-empty content').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hello\"}]}],\n \"inferenceConfig\": {\"maxTokens\": 256}\n}"}, "url": {"raw":"{{baseUrl}}/litellm/model/global.anthropic.claude-haiku-4-5-20251001-v1:0/converse","host":["{{baseUrl}}"],"path":["litellm", "model", "global.anthropic.claude-haiku-4-5-20251001-v1:0", "converse"]} } }, - { "name": "LiteLLM drop-in: Cohere shape chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Cohere shape: message.content non-empty', function () { var j = pm.response.json(); var c = j.message && j.message.content; pm.expect(Array.isArray(c) ? c.length > 0 : (typeof c === 'string' && c.length > 0), 'expected non-empty cohere message content').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"command-r-plus\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}"}, "url": {"raw":"{{baseUrl}}/litellm/v2/chat","host":["{{baseUrl}}"],"path":["litellm", "v2", "chat"]} } } - ] - }, - { - "name": "Cross-Cut Round 21: PydanticAI Drop-in Smoke", - "item": [ - { "name": "PydanticAI drop-in: OpenAI shape chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('OpenAI shape: choices[0].message.content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && (j.choices[0].message.content || (j.choices[0].message.tool_calls && j.choices[0].message.tool_calls.length))) || ''; pm.expect(c, 'no content or tool_calls').to.be.ok; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}"}, "url": {"raw":"{{baseUrl}}/pydanticai/v1/chat/completions","host":["{{baseUrl}}"],"path":["pydanticai", "v1", "chat", "completions"]} } }, - { "name": "PydanticAI drop-in: Anthropic shape chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic shape: content array non-empty', function () { var j = pm.response.json(); pm.expect(Array.isArray(j.content) && j.content.length > 0, 'expected non-empty content array').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}"}, "url": {"raw":"{{baseUrl}}/pydanticai/v1/messages","host":["{{baseUrl}}"],"path":["pydanticai", "v1", "messages"]} } }, - { "name": "PydanticAI drop-in: Gemini shape chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Gemini shape: candidates[0].content.parts non-empty', function () { var j = pm.response.json(); var parts = j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts; pm.expect(Array.isArray(parts) && parts.length > 0, 'expected non-empty parts').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Hello\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/pydanticai/v1beta/models/gemini-2.5-flash:generateContent","host":["{{baseUrl}}"],"path":["pydanticai", "v1beta", "models", "gemini-2.5-flash:generateContent"]} } }, - { "name": "PydanticAI drop-in: Bedrock shape converse", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Bedrock Converse shape: output.message.content non-empty', function () { var j = pm.response.json(); var content = j.output && j.output.message && j.output.message.content; pm.expect(Array.isArray(content) && content.length > 0, 'expected non-empty content').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hello\"}]}],\n \"inferenceConfig\": {\"maxTokens\": 256}\n}"}, "url": {"raw":"{{baseUrl}}/pydanticai/model/global.anthropic.claude-haiku-4-5-20251001-v1:0/converse","host":["{{baseUrl}}"],"path":["pydanticai", "model", "global.anthropic.claude-haiku-4-5-20251001-v1:0", "converse"]} } }, - { "name": "PydanticAI drop-in: Cohere shape chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Cohere shape: message.content non-empty', function () { var j = pm.response.json(); var c = j.message && j.message.content; pm.expect(Array.isArray(c) ? c.length > 0 : (typeof c === 'string' && c.length > 0), 'expected non-empty cohere message content').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"command-r-plus\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}"}, "url": {"raw":"{{baseUrl}}/pydanticai/v2/chat","host":["{{baseUrl}}"],"path":["pydanticai", "v2", "chat"]} } } - ] - }, - { - "name": "Cross-Cut Round 22: Cursor Drop-in Smoke", - "item": [ - { "name": "Cursor drop-in: OpenAI shape chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('OpenAI shape: choices[0].message.content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && (j.choices[0].message.content || (j.choices[0].message.tool_calls && j.choices[0].message.tool_calls.length))) || ''; pm.expect(c, 'no content or tool_calls').to.be.ok; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}"}, "url": {"raw":"{{baseUrl}}/cursor/v1/chat/completions","host":["{{baseUrl}}"],"path":["cursor", "v1", "chat", "completions"]} } }, - { "name": "Cursor drop-in: Anthropic shape chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic shape: content array non-empty', function () { var j = pm.response.json(); pm.expect(Array.isArray(j.content) && j.content.length > 0, 'expected non-empty content array').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}"}, "url": {"raw":"{{baseUrl}}/cursor/v1/messages","host":["{{baseUrl}}"],"path":["cursor", "v1", "messages"]} } }, - { "name": "Cursor drop-in: Gemini shape chat", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Gemini shape: candidates[0].content.parts non-empty', function () { var j = pm.response.json(); var parts = j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts; pm.expect(Array.isArray(parts) && parts.length > 0, 'expected non-empty parts').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Hello\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/cursor/v1beta/models/gemini-2.5-flash:generateContent","host":["{{baseUrl}}"],"path":["cursor", "v1beta", "models", "gemini-2.5-flash:generateContent"]} } }, - { "name": "Cursor drop-in: Bedrock shape converse", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Bedrock Converse shape: output.message.content non-empty', function () { var j = pm.response.json(); var content = j.output && j.output.message && j.output.message.content; pm.expect(Array.isArray(content) && content.length > 0, 'expected non-empty content').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Hello\"}]}],\n \"inferenceConfig\": {\"maxTokens\": 256}\n}"}, "url": {"raw":"{{baseUrl}}/cursor/model/global.anthropic.claude-haiku-4-5-20251001-v1:0/converse","host":["{{baseUrl}}"],"path":["cursor", "model", "global.anthropic.claude-haiku-4-5-20251001-v1:0", "converse"]} } } - ] - }, - { - "name": "Cross-Cut Round 23: Drop-in Structured Output Matrix (native shapes via /openai, /anthropic, /bedrock, /genai)", - "item": [ - { "name": "Drop-in /openai: gpt-5 (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: JSON with city/country/pop', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('not JSON: ' + e.message); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /openai: gpt-4o (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: JSON with city/country/pop', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('not JSON: ' + e.message); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /openai: gpt-4o-mini (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: JSON with city/country/pop', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('not JSON: ' + e.message); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /anthropic: claude-haiku-4-5 (forced tool emit_city)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic shape: content array non-empty', function () { var j = pm.response.json(); pm.expect(Array.isArray(j.content) && j.content.length > 0, 'expected non-empty content array').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"tools\": [{\"name\":\"emit_city\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"]}}],\n \"tool_choice\": {\"type\":\"tool\",\"name\":\"emit_city\"}\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic", "v1", "messages"]} } }, - { "name": "Drop-in /anthropic: claude-sonnet-4-6 (forced tool emit_city)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic shape: content array non-empty', function () { var j = pm.response.json(); pm.expect(Array.isArray(j.content) && j.content.length > 0, 'expected non-empty content array').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"tools\": [{\"name\":\"emit_city\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"]}}],\n \"tool_choice\": {\"type\":\"tool\",\"name\":\"emit_city\"}\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic", "v1", "messages"]} } }, - { "name": "Drop-in /bedrock: claude-haiku Converse (toolChoice emit_city)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Bedrock Converse shape: output.message.content non-empty', function () { var j = pm.response.json(); var content = j.output && j.output.message && j.output.message.content; pm.expect(Array.isArray(content) && content.length > 0, 'expected non-empty content').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Extract city/country/pop for Paris.\"}]}],\n \"toolConfig\": {\"tools\":[{\"toolSpec\":{\"name\":\"emit_city\",\"inputSchema\":{\"json\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"]}}}}],\"toolChoice\":{\"tool\":{\"name\":\"emit_city\"}}}\n}"}, "url": {"raw":"{{baseUrl}}/bedrock/model/global.anthropic.claude-haiku-4-5-20251001-v1:0/converse","host":["{{baseUrl}}"],"path":["bedrock", "model", "global.anthropic.claude-haiku-4-5-20251001-v1:0", "converse"]} } }, - { "name": "Drop-in /genai: gemini-2.5-flash (responseSchema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Gemini shape: candidates[0].content.parts non-empty', function () { var j = pm.response.json(); var parts = j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts; pm.expect(Array.isArray(parts) && parts.length > 0, 'expected non-empty parts').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Extract city/country/pop for Paris.\"}]}],\n \"generationConfig\": {\"responseMimeType\":\"application/json\",\"responseSchema\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"},\"country\":{\"type\":\"STRING\"},\"pop\":{\"type\":\"NUMBER\"}}}}\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/gemini-2.5-flash:generateContent","host":["{{baseUrl}}"],"path":["genai", "v1beta", "models", "gemini-2.5-flash:generateContent"]} } }, - { "name": "Drop-in /genai: gemini-2.5-pro (responseSchema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Gemini shape: candidates[0].content.parts non-empty', function () { var j = pm.response.json(); var parts = j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts; pm.expect(Array.isArray(parts) && parts.length > 0, 'expected non-empty parts').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Extract city/country/pop for Paris.\"}]}],\n \"generationConfig\": {\"responseMimeType\":\"application/json\",\"responseSchema\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"},\"country\":{\"type\":\"STRING\"},\"pop\":{\"type\":\"NUMBER\"}}}}\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/gemini-2.5-pro:generateContent","host":["{{baseUrl}}"],"path":["genai", "v1beta", "models", "gemini-2.5-pro:generateContent"]} } } - ] - }, - { - "name": "Cross-Cut Round 24: Drop-in Function Calling Matrix (native shapes)", - "item": [ - { "name": "Drop-in /openai: gpt-5 function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls && j.choices[0].message.tool_calls[0]; pm.expect(tc, 'no tool_calls').to.be.ok; pm.expect(tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /openai: gpt-4o function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls && j.choices[0].message.tool_calls[0]; pm.expect(tc, 'no tool_calls').to.be.ok; pm.expect(tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /openai: gpt-4o-mini function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls && j.choices[0].message.tool_calls[0]; pm.expect(tc, 'no tool_calls').to.be.ok; pm.expect(tc.function.name).to.equal('get_weather'); var a; try { a = JSON.parse(tc.function.arguments); } catch (e) { pm.expect.fail('args not JSON: ' + e.message); return; } pm.expect(a).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /anthropic: claude-opus-4-7 tool_use", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic: tool_use block with get_weather', function () { var j = pm.response.json(); var tu = (j.content || []).find(function (b) { return b.type === 'tool_use'; }); pm.expect(tu, 'no tool_use block').to.be.ok; pm.expect(tu.name).to.equal('get_weather'); pm.expect(tu.input).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"name\":\"get_weather\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic", "v1", "messages"]} } }, - { "name": "Drop-in /anthropic: claude-sonnet-4-6 tool_use", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic: tool_use block with get_weather', function () { var j = pm.response.json(); var tu = (j.content || []).find(function (b) { return b.type === 'tool_use'; }); pm.expect(tu, 'no tool_use block').to.be.ok; pm.expect(tu.name).to.equal('get_weather'); pm.expect(tu.input).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"name\":\"get_weather\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic", "v1", "messages"]} } }, - { "name": "Drop-in /anthropic: claude-haiku-4-5 tool_use", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic: tool_use block with get_weather', function () { var j = pm.response.json(); var tu = (j.content || []).find(function (b) { return b.type === 'tool_use'; }); pm.expect(tu, 'no tool_use block').to.be.ok; pm.expect(tu.name).to.equal('get_weather'); pm.expect(tu.input).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"name\":\"get_weather\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic", "v1", "messages"]} } }, - { "name": "Drop-in /bedrock: claude-sonnet-4-6 Converse tool_use", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Bedrock Converse shape: output.message.content non-empty', function () { var j = pm.response.json(); var content = j.output && j.output.message && j.output.message.content; pm.expect(Array.isArray(content) && content.length > 0, 'expected non-empty content').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Weather in Lagos, Nigeria?\"}]}],\n \"toolConfig\": {\"tools\":[{\"toolSpec\":{\"name\":\"get_weather\",\"inputSchema\":{\"json\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}}]}\n}"}, "url": {"raw":"{{baseUrl}}/bedrock/model/global.anthropic.claude-sonnet-4-6/converse","host":["{{baseUrl}}"],"path":["bedrock", "model", "global.anthropic.claude-sonnet-4-6", "converse"]} } }, - { "name": "Drop-in /genai: gemini-2.5-flash functionDeclarations", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Gemini: functionCall in parts', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var fc = parts.find(function (p) { return p && p.functionCall; }); pm.expect(fc, 'no functionCall').to.be.ok; pm.expect(fc.functionCall.name).to.equal('get_weather'); pm.expect(fc.functionCall.args).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Weather in Lagos, Nigeria?\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"get_weather\",\"parameters\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"}},\"required\":[\"city\"]}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/gemini-2.5-flash:generateContent","host":["{{baseUrl}}"],"path":["genai", "v1beta", "models", "gemini-2.5-flash:generateContent"]} } }, - { "name": "Drop-in /genai: gemini-2.5-pro functionDeclarations", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Gemini: functionCall in parts', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var fc = parts.find(function (p) { return p && p.functionCall; }); pm.expect(fc, 'no functionCall').to.be.ok; pm.expect(fc.functionCall.name).to.equal('get_weather'); pm.expect(fc.functionCall.args).to.have.property('city').that.is.a('string'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Weather in Lagos, Nigeria?\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"get_weather\",\"parameters\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"}},\"required\":[\"city\"]}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/gemini-2.5-pro:generateContent","host":["{{baseUrl}}"],"path":["genai", "v1beta", "models", "gemini-2.5-pro:generateContent"]} } } - ] - }, - { - "name": "Cross-Cut Round 25: Drop-in Vision Matrix (native shapes)", - "item": [ - { "name": "Drop-in /openai: gpt-5 vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /openai: gpt-4o vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /openai: gpt-4.1 vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4.1\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /anthropic: claude-opus-4-7 vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic", "v1", "messages"]} } }, - { "name": "Drop-in /anthropic: claude-sonnet-4-6 vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic", "v1", "messages"]} } }, - { "name": "Drop-in /anthropic: claude-haiku-4-5 vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic", "v1", "messages"]} } }, - { "name": "Drop-in /genai: gemini-2.5-flash vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Gemini vision: text part describes image', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (p) { return p && p.text; }); pm.expect(t, 'no text part').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/gemini-2.5-flash:generateContent","host":["{{baseUrl}}"],"path":["genai", "v1beta", "models", "gemini-2.5-flash:generateContent"]} } }, - { "name": "Drop-in /genai: gemini-2.5-pro vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Gemini vision: text part describes image', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (p) { return p && p.text; }); pm.expect(t, 'no text part').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/gemini-2.5-pro:generateContent","host":["{{baseUrl}}"],"path":["genai", "v1beta", "models", "gemini-2.5-pro:generateContent"]} } } - ] - }, - { - "name": "Cross-Cut Round 26: Drop-in Streaming Matrix (native shapes)", - "item": [ - { "name": "Drop-in /openai stream: gpt-4o", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /openai stream: gpt-4o-mini", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /openai stream: gpt-5", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-5\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/openai/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /anthropic stream: claude-opus-4-7", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-7\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic", "v1", "messages"]} } }, - { "name": "Drop-in /anthropic stream: claude-sonnet-4-6", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-sonnet-4-6\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic", "v1", "messages"]} } }, - { "name": "Drop-in /anthropic stream: claude-haiku-4-5", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic", "v1", "messages"]} } }, - { "name": "Drop-in /bedrock stream: claude-haiku Converse", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Bedrock stream: AWS event-stream or SSE', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected stream content-type, got ' + ct).to.match(/event-stream|vnd\\.amazon\\.eventstream/); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"text\":\"Count 1-5.\"}]}],\n \"inferenceConfig\": {\"maxTokens\": 256}\n}"}, "url": {"raw":"{{baseUrl}}/bedrock/model/global.anthropic.claude-haiku-4-5-20251001-v1:0/converse-stream","host":["{{baseUrl}}"],"path":["bedrock", "model", "global.anthropic.claude-haiku-4-5-20251001-v1:0", "converse-stream"]} } }, - { "name": "Drop-in /genai stream: gemini-2.5-flash", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Count 1-5.\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/gemini-2.5-flash:streamGenerateContent?alt=sse","host":["{{baseUrl}}"],"path":["genai", "v1beta", "models", "gemini-2.5-flash:streamGenerateContent"],"query":[{"key":"alt","value":"sse"}]} } }, - { "name": "Drop-in /genai stream: gemini-2.5-pro", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Count 1-5.\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai/v1beta/models/gemini-2.5-pro:streamGenerateContent?alt=sse","host":["{{baseUrl}}"],"path":["genai", "v1beta", "models", "gemini-2.5-pro:streamGenerateContent"],"query":[{"key":"alt","value":"sse"}]} } } - ] - }, - { - "name": "Cross-Cut Round 27: Drop-in Umbrella Vision Matrix (vision via /langchain, /litellm, /pydanticai, /cursor)", - "item": [ - { "name": "Drop-in /langchain vision: OpenAI shape", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/langchain/v1/chat/completions","host":["{{baseUrl}}"],"path":["langchain", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /langchain vision: Anthropic shape", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/langchain/v1/messages","host":["{{baseUrl}}"],"path":["langchain", "v1", "messages"]} } }, - { "name": "Drop-in /langchain vision: Gemini shape", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: text part describes image', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (p) { return p && p.text; }); pm.expect(t, 'no text part').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/langchain/v1beta/models/gemini-2.5-flash:generateContent","host":["{{baseUrl}}"],"path":["langchain", "v1beta", "models", "gemini-2.5-flash:generateContent"]} } }, - { "name": "Drop-in /litellm vision: OpenAI shape", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/litellm/v1/chat/completions","host":["{{baseUrl}}"],"path":["litellm", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /litellm vision: Anthropic shape", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/litellm/v1/messages","host":["{{baseUrl}}"],"path":["litellm", "v1", "messages"]} } }, - { "name": "Drop-in /litellm vision: Gemini shape", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: text part describes image', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (p) { return p && p.text; }); pm.expect(t, 'no text part').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/litellm/v1beta/models/gemini-2.5-flash:generateContent","host":["{{baseUrl}}"],"path":["litellm", "v1beta", "models", "gemini-2.5-flash:generateContent"]} } }, - { "name": "Drop-in /pydanticai vision: OpenAI shape", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/pydanticai/v1/chat/completions","host":["{{baseUrl}}"],"path":["pydanticai", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /pydanticai vision: Anthropic shape", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/pydanticai/v1/messages","host":["{{baseUrl}}"],"path":["pydanticai", "v1", "messages"]} } }, - { "name": "Drop-in /pydanticai vision: Gemini shape", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: text part describes image', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (p) { return p && p.text; }); pm.expect(t, 'no text part').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/pydanticai/v1beta/models/gemini-2.5-flash:generateContent","host":["{{baseUrl}}"],"path":["pydanticai", "v1beta", "models", "gemini-2.5-flash:generateContent"]} } }, - { "name": "Drop-in /cursor vision: OpenAI shape", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o\",\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/cursor/v1/chat/completions","host":["{{baseUrl}}"],"path":["cursor", "v1", "chat", "completions"]} } }, - { "name": "Drop-in /cursor vision: Anthropic shape", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/cursor/v1/messages","host":["{{baseUrl}}"],"path":["cursor", "v1", "messages"]} } }, - { "name": "Drop-in /cursor vision: Gemini shape", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: text part describes image', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (p) { return p && p.text; }); pm.expect(t, 'no text part').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/cursor/v1beta/models/gemini-2.5-flash:generateContent","host":["{{baseUrl}}"],"path":["cursor", "v1beta", "models", "gemini-2.5-flash:generateContent"]} } } - ] - }, - { - "name": "Cross-Cut Round 28: Passthrough Advanced Matrix (features via *_passthrough byte-for-byte routes)", - "item": [ - { "name": "Passthrough /openai: structured output (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: JSON with city/country/pop', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('not JSON: ' + e.message); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/openai_passthrough/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai_passthrough", "v1", "chat", "completions"]} } }, - { "name": "Passthrough /openai: function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Function call: get_weather invoked', function () { var j = pm.response.json(); var tc = j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.tool_calls && j.choices[0].message.tool_calls[0]; pm.expect(tc, 'no tool_calls').to.be.ok; pm.expect(tc.function.name).to.equal('get_weather'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]\n}"}, "url": {"raw":"{{baseUrl}}/openai_passthrough/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai_passthrough", "v1", "chat", "completions"]} } }, - { "name": "Passthrough /openai: streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"Authorization","value":"Bearer {{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"model\": \"gpt-4o-mini\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/openai_passthrough/v1/chat/completions","host":["{{baseUrl}}"],"path":["openai_passthrough", "v1", "chat", "completions"]} } }, - { "name": "Passthrough /anthropic: function calling (tool_use)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic: tool_use block', function () { var j = pm.response.json(); var tu = (j.content || []).find(function (b) { return b.type === 'tool_use'; }); pm.expect(tu, 'no tool_use').to.be.ok; pm.expect(tu.name).to.equal('get_weather'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Weather in Lagos, Nigeria?\"}],\n \"tools\": [{\"name\":\"get_weather\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic_passthrough/v1/messages","host":["{{baseUrl}}"],"path":["anthropic_passthrough", "v1", "messages"]} } }, - { "name": "Passthrough /anthropic: vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: text block describes image', function () { var j = pm.response.json(); var t = (j.content || []).find(function (b) { return b.type === 'text' && b.text; }); pm.expect(t, 'no text block').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 512,\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"image\",\"source\":{\"type\":\"url\",\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}},{\"type\":\"text\",\"text\":\"Describe\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic_passthrough/v1/messages","host":["{{baseUrl}}"],"path":["anthropic_passthrough", "v1", "messages"]} } }, - { "name": "Passthrough /anthropic: multi-turn", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Multi-turn: response present', function () { var j = pm.response.json(); var c = (j.content && j.content.length > 0) || (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content); pm.expect(c, 'no content').to.be.ok; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"},{\"role\":\"assistant\",\"content\":\"Hello\"},{\"role\":\"user\",\"content\":\"How are you?\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic_passthrough/v1/messages","host":["{{baseUrl}}"],"path":["anthropic_passthrough", "v1", "messages"]} } }, - { "name": "Passthrough /anthropic: streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-haiku-4-5\",\n \"max_tokens\": 800,\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/anthropic_passthrough/v1/messages","host":["{{baseUrl}}"],"path":["anthropic_passthrough", "v1", "messages"]} } }, - { "name": "Passthrough /genai: function calling", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Gemini: functionCall present', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var fc = parts.find(function (p) { return p && p.functionCall; }); pm.expect(fc, 'no functionCall').to.be.ok; pm.expect(fc.functionCall.name).to.equal('get_weather'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Weather in Lagos, Nigeria?\"}]}],\n \"tools\": [{\"functionDeclarations\":[{\"name\":\"get_weather\",\"parameters\":{\"type\":\"OBJECT\",\"properties\":{\"city\":{\"type\":\"STRING\"}},\"required\":[\"city\"]}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai_passthrough/v1beta/models/gemini-2.5-flash:generateContent","host":["{{baseUrl}}"],"path":["genai_passthrough", "v1beta", "models", "gemini-2.5-flash:generateContent"]} } }, - { "name": "Passthrough /genai: vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: text part describes image', function () { var j = pm.response.json(); var parts = (j.candidates && j.candidates[0] && j.candidates[0].content && j.candidates[0].content.parts) || []; var t = parts.find(function (p) { return p && p.text; }); pm.expect(t, 'no text part').to.be.ok; pm.expect(t.text.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Describe\"},{\"fileData\":{\"mimeType\":\"image/jpeg\",\"fileUri\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai_passthrough/v1beta/models/gemini-2.5-flash:generateContent","host":["{{baseUrl}}"],"path":["genai_passthrough", "v1beta", "models", "gemini-2.5-flash:generateContent"]} } }, - { "name": "Passthrough /genai: streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-goog-api-key","value":"{{genaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"contents\": [{\"parts\":[{\"text\":\"Count 1-5.\"}]}]\n}"}, "url": {"raw":"{{baseUrl}}/genai_passthrough/v1beta/models/gemini-2.5-flash:streamGenerateContent?alt=sse","host":["{{baseUrl}}"],"path":["genai_passthrough", "v1beta", "models", "gemini-2.5-flash:streamGenerateContent"],"query":[{"key":"alt","value":"sse"}]} } }, - { "name": "Passthrough /azure: structured output (json_schema)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Structured output: JSON with city/country/pop', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; var p; try { p = JSON.parse(c); } catch (e) { pm.expect.fail('not JSON: ' + e.message); return; } pm.expect(p).to.have.property('city').that.is.a('string'); pm.expect(p).to.have.property('country').that.is.a('string'); pm.expect(p).to.have.property('pop').that.is.a('number'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":\"Extract city/country/pop for Paris.\"}],\n \"response_format\": {\"type\":\"json_schema\",\"json_schema\":{\"name\":\"city\",\"strict\":true,\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"},\"country\":{\"type\":\"string\"},\"pop\":{\"type\":\"number\"}},\"required\":[\"city\",\"country\",\"pop\"],\"additionalProperties\":false}}}\n}"}, "url": {"raw":"{{baseUrl}}/azure_passthrough/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["azure_passthrough", "openai", "deployments", "{{azureDeployment}}", "chat", "completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]} } }, - { "name": "Passthrough /azure: vision", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Vision: response describes image', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; pm.expect(c.length).to.be.greaterThan(20); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Describe\"},{\"type\":\"image_url\",\"image_url\":{\"url\":\"https://storage.googleapis.com/generativeai-downloads/images/scones.jpg\"}}]}]\n}"}, "url": {"raw":"{{baseUrl}}/azure_passthrough/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["azure_passthrough", "openai", "deployments", "{{azureDeployment}}", "chat", "completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]} } }, - { "name": "Passthrough /azure: streaming", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Streaming: SSE response', function () { var ct = pm.response.headers.get('content-type') || ''; pm.expect(ct, 'expected event-stream, got ' + ct).to.include('event-stream'); }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"api-key","value":"{{openaiKey}}"}], "body": {"mode":"raw","raw":"{\n \"messages\": [{\"role\":\"user\",\"content\":\"Count 1-5.\"}],\n \"stream\": true\n}"}, "url": {"raw":"{{baseUrl}}/azure_passthrough/openai/deployments/{{azureDeployment}}/chat/completions?api-version={{azureApiVersion}}","host":["{{baseUrl}}"],"path":["azure_passthrough", "openai", "deployments", "{{azureDeployment}}", "chat", "completions"],"query":[{"key":"api-version","value":"{{azureApiVersion}}"}]} } } - ] - }, - { - "name": "Cross-Cut Round 29: Mid-Conversation System Message Matrix", - "description": "Verifies mid-conversation role:system handling across providers.\n- Anthropic + Opus 4.8: system emitted as role:\"system\" in messages array (native support, no beta header).\n- Bedrock + Vertex + Opus 4.7: system content merged into top-level system field (fallback — positional semantics lost, data preserved).\nPlacement rule: system must end the array OR be immediately followed by an assistant turn.", - "item": [ - { "name": "Cross-cut: anthropic/claude-opus-4-8 mid-conv system (ends array)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"},\n {\"role\":\"system\",\"content\":\"Respond only in one word.\"}\n ]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-8 mid-conv system (before assistant)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"system\",\"content\":\"From now on be very concise.\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"}\n ]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: bedrock/claude-opus-4-8 mid-conv system (fallback: merged to top-level)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"bedrock/global.anthropic.claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"},\n {\"role\":\"system\",\"content\":\"Respond only in one word.\"}\n ]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: vertex/claude-opus-4-8 mid-conv system (fallback: merged to top-level)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"vertex/claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"},\n {\"role\":\"system\",\"content\":\"Respond only in one word.\"}\n ]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-7 mid-conv system (fallback: merged to top-level)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-7\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"},\n {\"role\":\"system\",\"content\":\"Respond only in one word.\"}\n ]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-8 mid-conv system drop-in /anthropic (ends array)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 512,\n \"system\": [{\"type\":\"text\",\"text\":\"You are a helpful assistant.\"}],\n \"messages\": [\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Hello\"}]},\n {\"role\":\"assistant\",\"content\":[{\"type\":\"text\",\"text\":\"Hi!\"}]},\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"How are you?\"}]},\n {\"role\":\"system\",\"content\":[{\"type\":\"text\",\"text\":\"Respond only in one word.\"}]}\n ]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}} - ] - }, - { - "name": "Cross-Cut Round 30: Opus 4.8 Feature Gating", - "description": "Exercises the Anthropic request-surface features on Claude Opus 4.8, the live successor after Fable 5 / Mythos were discontinued (https://www.anthropic.com/news/fable-mythos-access).\nSurface covered: adaptive thinking, budget_tokens to adaptive, sampling-param handling (temperature/top_p/top_k), effort high/xhigh/max, structured outputs, task budgets, computer-use new-gen tools, web_search dynamic filtering, mid-conversation system messages, and fast mode.\nScripts are guarded on code<400 so the suite tolerates accounts without access to a given feature (same convention as the [PREVIEW] items).", - "item": [ - { "name": "Native Opus 4.8: adaptive thinking", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": { \"type\": \"adaptive\" },\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Native Opus 4.8: output_config effort high", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 1024,\n \"output_config\": {\"effort\": \"high\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve x^2 - 5x + 6 = 0\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Native Opus 4.8: output_config effort xhigh", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 1024,\n \"output_config\": {\"effort\": \"xhigh\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve x^2 - 5x + 6 = 0\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Native Opus 4.8: output_config effort max", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 1024,\n \"output_config\": {\"effort\": \"max\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve x^2 - 5x + 6 = 0\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Native Opus 4.8: output_config format json_schema (structured outputs)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 800,\n \"output_config\": {\"format\": {\"type\":\"json_schema\",\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}}}}},\n \"messages\": [{\"role\":\"user\",\"content\":\"Pick a city\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Native Opus 4.8: speed:fast stripped (fast mode unsupported, no beta header)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 800,\n \"speed\": \"fast\",\n \"messages\": [{\"role\":\"user\",\"content\":\"Hi\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Native Opus 4.8: task_budget (beta task-budgets-2026-03-13)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"task-budgets-2026-03-13"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"output_config\": {\"task_budget\": {\"type\":\"tokens\",\"total\":20000}},\n \"messages\": [{\"role\":\"user\",\"content\":\"Plan and solve a multi-step task.\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Native Opus 4.8: computer use new-gen tools (computer-use-2025-11-24)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"},{"key":"anthropic-beta","value":"computer-use-2025-11-24"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"tools\": [\n { \"type\": \"computer_20251124\", \"name\": \"computer\", \"display_width_px\": 1024, \"display_height_px\": 768, \"display_number\": 1 },\n { \"type\": \"bash_20250124\", \"name\": \"bash\" },\n { \"type\": \"text_editor_20250728\", \"name\": \"str_replace_based_edit_tool\" }\n ],\n \"messages\": [{\"role\":\"user\",\"content\":\"Take a screenshot of the desktop.\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Native Opus 4.8: web_search dynamic filtering (web_search_20260209)", "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 1024,\n \"messages\": [{\"role\":\"user\",\"content\":\"Find recent AI papers.\"}],\n \"tools\": [{\"type\":\"web_search_20260209\",\"name\":\"web_search\",\"max_uses\":3}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Native Opus 4.8: mid-conv system message (ends array)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 512,\n \"system\": [{\"type\":\"text\",\"text\":\"You are a helpful assistant.\"}],\n \"messages\": [\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Hello\"}]},\n {\"role\":\"assistant\",\"content\":[{\"type\":\"text\",\"text\":\"Hi!\"}]},\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"How are you?\"}]},\n {\"role\":\"system\",\"content\":[{\"type\":\"text\",\"text\":\"Respond only in one word.\"}]}\n ]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Native Opus 4.8: mid-conv system message (before assistant)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 512,\n \"system\": [{\"type\":\"text\",\"text\":\"You are a helpful assistant.\"}],\n \"messages\": [\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Hello\"}]},\n {\"role\":\"system\",\"content\":[{\"type\":\"text\",\"text\":\"From now on be very concise.\"}]},\n {\"role\":\"assistant\",\"content\":[{\"type\":\"text\",\"text\":\"Hi!\"}]},\n {\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"How are you?\"}]}\n ]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Native Opus 4.8: adaptive thinking (family parity)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Anthropic drop-in: content block present', function () { var j = pm.response.json(); var hasContent = Array.isArray(j.content) && j.content.some(function(b) { return b.type === 'text' && b.text; }); pm.expect(hasContent, 'expected text content block').to.be.true; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"},{"key":"x-api-key","value":"{{anthropicKey}}"},{"key":"anthropic-version","value":"2023-06-01"}], "body": {"mode":"raw","raw":"{\n \"model\": \"claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": { \"type\": \"adaptive\" },\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/anthropic/v1/messages","host":["{{baseUrl}}"],"path":["anthropic","v1","messages"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-8 adaptive thinking", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"adaptive\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-8 enabled thinking → adaptive (budget_tokens removed)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 4096,\n \"thinking\": {\"type\":\"enabled\",\"budget_tokens\":2000},\n \"messages\": [{\"role\":\"user\",\"content\":\"Solve in steps\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-8 disabled thinking → omitted (no 400)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 1024,\n \"thinking\": {\"type\":\"disabled\"},\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-8 sampling params stripped (temperature/top_p/top_k)", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 1024,\n \"temperature\": 0.7,\n \"top_p\": 0.9,\n \"top_k\": 5,\n \"messages\": [{\"role\":\"user\",\"content\":\"Hello\"}]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}}, - { "name": "Cross-cut: anthropic/claude-opus-4-8 mid-conv system", "event": [{"listen":"test","script":{"type":"text/javascript","exec":["if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }"]}}], "request": { "method": "POST", "header": [{"key":"Content-Type","value":"application/json"}], "body": {"mode":"raw","raw":"{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"},\n {\"role\":\"system\",\"content\":\"Respond only in one word.\"}\n ]\n}"}, "url": {"raw":"{{baseUrl}}/v1/chat/completions","host":["{{baseUrl}}"],"path":["v1","chat","completions"]}}} + }, + { + "name": "Cross-cut: anthropic/claude-opus-4-8 mid-conv system", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Response content non-empty', function () { var j = pm.response.json(); var c = (j.choices && j.choices[0] && j.choices[0].message && j.choices[0].message.content) || ''; pm.expect(c).to.be.a('string').and.not.empty; }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-opus-4-8\",\n \"max_tokens\": 512,\n \"messages\": [\n {\"role\":\"system\",\"content\":\"You are a helpful assistant.\"},\n {\"role\":\"user\",\"content\":\"Hello\"},\n {\"role\":\"assistant\",\"content\":\"Hi!\"},\n {\"role\":\"user\",\"content\":\"How are you?\"},\n {\"role\":\"system\",\"content\":\"Respond only in one word.\"}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + } + } ] } - ] }, { @@ -8075,180 +35676,743 @@ } ] } - ] - }, - { - "name": "11. Cross-Provider Feature Tests", - "description": "Same feature exercised across multiple providers via Bifrost's unified routing. Each sub-folder is one capability; each request differs only by `model` (or path) to show that Bifrost translates the request shape per-provider.", - "item": [ + ] + }, + { + "name": "11. Cross-Provider Feature Tests", + "description": "Same feature exercised across multiple providers via Bifrost's unified routing. Each sub-folder is one capability; each request differs only by `model` (or path) to show that Bifrost translates the request shape per-provider.", + "item": [ + { + "name": "Context Compaction cross-cut", + "item": [ + { + "name": "Compaction (Azure gpt-4o)", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code < 400) { pm.test('Compaction: object is response.compaction', function () { var j = pm.response.json(); pm.expect(j.object).to.equal('response.compaction'); }); pm.test('Compaction: output is non-empty array', function () { var j = pm.response.json(); pm.expect(j.output).to.be.an('array').and.not.empty; }); pm.test('Compaction: last output item has type response.compaction and encrypted_content', function () { var j = pm.response.json(); var last = j.output[j.output.length - 1]; pm.expect(last.type).to.equal('compaction'); pm.expect(last.encrypted_content).to.be.a('string').and.not.empty; }); pm.test('Compaction: usage is present', function () { var j = pm.response.json(); pm.expect(j.usage).to.be.an('object'); pm.expect(j.usage.input_tokens).to.be.a('number'); }); }" + ] + } + } + ], + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "Authorization", + "value": "Bearer {{openaiKey}}" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"{{azureDeployment}}\",\n \"input\": [\n {\"role\": \"user\", \"content\": \"What is the capital of France?\"},\n {\"role\": \"assistant\", \"content\": \"The capital of France is Paris.\"},\n {\"role\": \"user\", \"content\": \"What is the population of Paris?\"},\n {\"role\": \"assistant\", \"content\": \"Paris has a population of approximately 2.1 million in the city proper, and around 12 million in the greater metropolitan area.\"},\n {\"role\": \"user\", \"content\": \"What is Paris known for?\"},\n {\"role\": \"assistant\", \"content\": \"Paris is known for the Eiffel Tower, the Louvre Museum, Notre-Dame Cathedral, world-class cuisine, fashion, and its rich history as a cultural and political center of Europe.\"}\n ]\n}" + }, + "url": { + "raw": "{{baseUrl}}/openai/v1/responses/compact", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "openai", + "v1", + "responses", + "compact" + ] + } + } + } + ] + } + ] + } + ] + }, + { + "name": "15. Costing — Failed/Cancelled Requests", + "description": "Asserts that a request which FAILED or was CANCELLED but still consumed provider tokens is billed (logs DB cost>0 && total_tokens>0). Each request pins a unique `x-request-id` and sets `x-bf-expect-cost: true`, which the newman-reporter-dbverify reporter uses to look up the log row and verify cost. Pre-fix these FAIL (cost is 0 on failed requests); post-fix they PASS. Streaming mid-flight cancellation is driven deterministically by runners/run-stream-cancellation.mjs (Postman cannot abort a stream mid-response).", + "item": [ + { + "name": "[Costing] Streaming chat cancelled mid-flight records cost", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-request-id", + "value": "costing-stream-chat-{{$guid}}" + }, + { + "key": "x-bf-expect-cost", + "value": "true" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"{{model}}\",\n \"messages\": [{\"role\": \"user\", \"content\": \"Count slowly from 1 to 200, one number per line.\"}],\n \"stream\": true,\n \"max_tokens\": 2048\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Stream returns 200 (headers sent) then is cancelled mid-response by", + "// runners/run-stream-cancellation.mjs. The cost assertion is performed by", + "// the dbverify reporter against the logs row keyed on x-request-id.", + "pm.test('[Costing] stream started (200) before cancellation', function () {", + " pm.expect(pm.response.code).to.equal(200);", + "});" + ] + } + } + ] + }, + { + "name": "[Costing] Non-streaming chat failure records cost", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-request-id", + "value": "costing-nonstream-chat-{{$guid}}" + }, + { + "key": "x-bf-expect-cost", + "value": "true" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"{{model}}\",\n \"messages\": [{\"role\": \"user\", \"content\": \"Write a long essay about distributed systems.\"}],\n \"max_tokens\": 1024\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// A non-streaming request that fails AFTER the provider processed input", + "// tokens (timeout / 5xx after input) must still be billed. The dbverify", + "// reporter verifies logs cost>0 && total_tokens>0 for x-request-id.", + "pm.test('[Costing] non-stream request reached the provider', function () {", + " pm.expect(pm.response.code).to.not.equal(404);", + "});" + ] + } + } + ] + }, + { + "name": "[Costing] Embeddings failure records cost", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-request-id", + "value": "costing-embeddings-{{$guid}}" + }, + { + "key": "x-bf-expect-cost", + "value": "true" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"{{embedding_model}}\",\n \"input\": \"costing probe embedding input\"\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/embeddings", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "embeddings" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// 'Other' request type: embeddings. A failure after input processing must", + "// still record cost. Verified by the dbverify reporter via x-request-id.", + "pm.test('[Costing] embeddings request reached the provider', function () {", + " pm.expect(pm.response.code).to.not.equal(404);", + "});" + ] + } + } + ] + }, + { + "name": "[Costing] Transcription failure records cost", + "request": { + "method": "POST", + "header": [ + { + "key": "x-request-id", + "value": "costing-transcription-{{$guid}}" + }, + { + "key": "x-bf-expect-cost", + "value": "true" + } + ], + "body": { + "mode": "formdata", + "formdata": [ + { + "key": "model", + "value": "{{transcription_model}}", + "type": "text" + }, + { + "key": "file", + "src": "fixtures/sample.mp3", + "type": "file" + } + ] + }, + "url": { + "raw": "{{baseUrl}}/v1/audio/transcriptions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "audio", + "transcriptions" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// 'Other' request type: audio transcription. A failure after the audio was", + "// processed must still record cost. Verified by the dbverify reporter.", + "pm.test('[Costing] transcription request reached the provider', function () {", + " pm.expect(pm.response.code).to.not.equal(404);", + "});" + ] + } + } + ] + } + ] + }, + { + "name": "16. Accounting (cost recording across providers)", + "description": "Criss-cross of providers x {non-streaming, streaming} chat requests that each pin a unique x-request-id and set x-bf-expect-cost. The dbverify reporter asserts the logs DB recorded cost>0 && total_tokens>0 per request, guarding against accounting regressions. Failed-streaming, retry-attempt, cumulative-sum and no-double-bill semantics are covered by Go unit tests (plugins/governance/accounting_test.go).", + "item": [ + { + "name": "[Accounting] openai non-streaming records cost", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-request-id", + "value": "acct-openai-nonstreaming" + }, + { + "key": "x-bf-expect-cost", + "value": "true" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Cost recording is verified against the logs DB by the dbverify reporter", + "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", + "// gateway accepted and routed it.", + "pm.test(`[Accounting] openai nonstreaming reached the provider`, function () {", + " pm.expect(pm.response.code).to.not.equal(404);", + "});" + ] + } + } + ] + }, + { + "name": "[Accounting] openai streaming records cost", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-request-id", + "value": "acct-openai-streaming" + }, + { + "key": "x-bf-expect-cost", + "value": "true" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512,\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Cost recording is verified against the logs DB by the dbverify reporter", + "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", + "// gateway accepted and routed it.", + "pm.test(`[Accounting] openai streaming reached the provider`, function () {", + " pm.expect(pm.response.code).to.not.equal(404);", + "});" + ] + } + } + ] + }, + { + "name": "[Accounting] anthropic non-streaming records cost", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-request-id", + "value": "acct-anthropic-nonstreaming" + }, + { + "key": "x-bf-expect-cost", + "value": "true" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Cost recording is verified against the logs DB by the dbverify reporter", + "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", + "// gateway accepted and routed it.", + "pm.test(`[Accounting] anthropic nonstreaming reached the provider`, function () {", + " pm.expect(pm.response.code).to.not.equal(404);", + "});" + ] + } + } + ] + }, + { + "name": "[Accounting] anthropic streaming records cost", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-request-id", + "value": "acct-anthropic-streaming" + }, + { + "key": "x-bf-expect-cost", + "value": "true" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512,\n \"stream\": true\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + }, + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Cost recording is verified against the logs DB by the dbverify reporter", + "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", + "// gateway accepted and routed it.", + "pm.test(`[Accounting] anthropic streaming reached the provider`, function () {", + " pm.expect(pm.response.code).to.not.equal(404);", + "});" + ] + } + } + ] + }, + { + "name": "[Accounting] bedrock non-streaming records cost", + "request": { + "method": "POST", + "header": [ + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-request-id", + "value": "acct-bedrock-nonstreaming" + }, + { + "key": "x-bf-expect-cost", + "value": "true" + } + ], + "body": { + "mode": "raw", + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512\n}" + }, + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } + }, + "event": [ { - "name": "Context Compaction cross-cut", - "item": [ - { - "name": "Compaction (Azure gpt-4o)", - "event": [ - { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "if (pm.response.code < 400) { pm.test('Compaction: object is response.compaction', function () { var j = pm.response.json(); pm.expect(j.object).to.equal('response.compaction'); }); pm.test('Compaction: output is non-empty array', function () { var j = pm.response.json(); pm.expect(j.output).to.be.an('array').and.not.empty; }); pm.test('Compaction: last output item has type response.compaction and encrypted_content', function () { var j = pm.response.json(); var last = j.output[j.output.length - 1]; pm.expect(last.type).to.equal('compaction'); pm.expect(last.encrypted_content).to.be.a('string').and.not.empty; }); pm.test('Compaction: usage is present', function () { var j = pm.response.json(); pm.expect(j.usage).to.be.an('object'); pm.expect(j.usage.input_tokens).to.be.a('number'); }); }" - ] - } - } - ], - "request": { - "method": "POST", - "header": [ - { - "key": "Content-Type", - "value": "application/json" - }, - { - "key": "Authorization", - "value": "Bearer {{openaiKey}}" - } - ], - "body": { - "mode": "raw", - "raw": "{\n \"model\": \"{{azureDeployment}}\",\n \"input\": [\n {\"role\": \"user\", \"content\": \"What is the capital of France?\"},\n {\"role\": \"assistant\", \"content\": \"The capital of France is Paris.\"},\n {\"role\": \"user\", \"content\": \"What is the population of Paris?\"},\n {\"role\": \"assistant\", \"content\": \"Paris has a population of approximately 2.1 million in the city proper, and around 12 million in the greater metropolitan area.\"},\n {\"role\": \"user\", \"content\": \"What is Paris known for?\"},\n {\"role\": \"assistant\", \"content\": \"Paris is known for the Eiffel Tower, the Louvre Museum, Notre-Dame Cathedral, world-class cuisine, fashion, and its rich history as a cultural and political center of Europe.\"}\n ]\n}" - }, - "url": { - "raw": "{{baseUrl}}/openai/v1/responses/compact", - "host": [ - "{{baseUrl}}" - ], - "path": [ - "openai", - "v1", - "responses", - "compact" - ] - } - } - } - ] + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Cost recording is verified against the logs DB by the dbverify reporter", + "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", + "// gateway accepted and routed it.", + "pm.test(`[Accounting] bedrock nonstreaming reached the provider`, function () {", + " pm.expect(pm.response.code).to.not.equal(404);", + "});" + ] + } } ] - } - ] - }, - { - "name": "15. Costing — Failed/Cancelled Requests", - "description": "Asserts that a request which FAILED or was CANCELLED but still consumed provider tokens is billed (logs DB cost>0 && total_tokens>0). Each request pins a unique `x-request-id` and sets `x-bf-expect-cost: true`, which the newman-reporter-dbverify reporter uses to look up the log row and verify cost. Pre-fix these FAIL (cost is 0 on failed requests); post-fix they PASS. Streaming mid-flight cancellation is driven deterministically by runners/run-stream-cancellation.mjs (Postman cannot abort a stream mid-response).", - "item": [ + }, { - "name": "[Costing] Streaming chat cancelled mid-flight records cost", + "name": "[Accounting] bedrock non-streaming records cost", "request": { "method": "POST", "header": [ - { "key": "Content-Type", "value": "application/json" }, - { "key": "x-request-id", "value": "costing-stream-chat-{{$guid}}" }, - { "key": "x-bf-expect-cost", "value": "true" } + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-request-id", + "value": "acct-bedrock-nonstreaming" + }, + { + "key": "x-bf-expect-cost", + "value": "true" + } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"{{model}}\",\n \"messages\": [{\"role\": \"user\", \"content\": \"Count slowly from 1 to 200, one number per line.\"}],\n \"stream\": true,\n \"max_tokens\": 2048\n}" + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512\n}" }, - "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } }, "event": [ - { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// Stream returns 200 (headers sent) then is cancelled mid-response by", - "// runners/run-stream-cancellation.mjs. The cost assertion is performed by", - "// the dbverify reporter against the logs row keyed on x-request-id.", - "pm.test('[Costing] stream started (200) before cancellation', function () {", - " pm.expect(pm.response.code).to.equal(200);", - "});" - ] } } + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Cost recording is verified against the logs DB by the dbverify reporter", + "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", + "// gateway accepted and routed it.", + "pm.test(`[Accounting] bedrock nonstreaming reached the provider`, function () {", + " pm.expect(pm.response.code).to.not.equal(404);", + "});" + ] + } + } ] }, { - "name": "[Costing] Non-streaming chat failure records cost", + "name": "[Accounting] bedrock streaming records cost", "request": { "method": "POST", "header": [ - { "key": "Content-Type", "value": "application/json" }, - { "key": "x-request-id", "value": "costing-nonstream-chat-{{$guid}}" }, - { "key": "x-bf-expect-cost", "value": "true" } + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-request-id", + "value": "acct-bedrock-streaming" + }, + { + "key": "x-bf-expect-cost", + "value": "true" + } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"{{model}}\",\n \"messages\": [{\"role\": \"user\", \"content\": \"Write a long essay about distributed systems.\"}],\n \"max_tokens\": 1024\n}" + "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512,\n \"stream\": true\n}" }, - "url": { "raw": "{{baseUrl}}/v1/chat/completions", "host": ["{{baseUrl}}"], "path": ["v1", "chat", "completions"] } + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } }, "event": [ - { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// A non-streaming request that fails AFTER the provider processed input", - "// tokens (timeout / 5xx after input) must still be billed. The dbverify", - "// reporter verifies logs cost>0 && total_tokens>0 for x-request-id.", - "pm.test('[Costing] non-stream request reached the provider', function () {", - " pm.expect(pm.response.code).to.not.equal(404);", - "});" - ] } } + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Cost recording is verified against the logs DB by the dbverify reporter", + "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", + "// gateway accepted and routed it.", + "pm.test(`[Accounting] bedrock streaming reached the provider`, function () {", + " pm.expect(pm.response.code).to.not.equal(404);", + "});" + ] + } + } ] }, { - "name": "[Costing] Embeddings failure records cost", + "name": "[Accounting] bedrock streaming records cost", "request": { "method": "POST", "header": [ - { "key": "Content-Type", "value": "application/json" }, - { "key": "x-request-id", "value": "costing-embeddings-{{$guid}}" }, - { "key": "x-bf-expect-cost", "value": "true" } + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-request-id", + "value": "acct-bedrock-streaming" + }, + { + "key": "x-bf-expect-cost", + "value": "true" + } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"{{embedding_model}}\",\n \"input\": \"costing probe embedding input\"\n}" + "raw": "{\n \"model\": \"bedrock_mantle/anthropic.claude-opus-4-8\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512,\n \"stream\": true\n}" }, - "url": { "raw": "{{baseUrl}}/v1/embeddings", "host": ["{{baseUrl}}"], "path": ["v1", "embeddings"] } + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } }, "event": [ - { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// 'Other' request type: embeddings. A failure after input processing must", - "// still record cost. Verified by the dbverify reporter via x-request-id.", - "pm.test('[Costing] embeddings request reached the provider', function () {", - " pm.expect(pm.response.code).to.not.equal(404);", - "});" - ] } } + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Cost recording is verified against the logs DB by the dbverify reporter", + "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", + "// gateway accepted and routed it.", + "pm.test(`[Accounting] bedrock streaming reached the provider`, function () {", + " pm.expect(pm.response.code).to.not.equal(404);", + "});" + ] + } + } ] }, { - "name": "[Costing] Transcription failure records cost", + "name": "[Accounting] gemini non-streaming records cost", "request": { "method": "POST", "header": [ - { "key": "x-request-id", "value": "costing-transcription-{{$guid}}" }, - { "key": "x-bf-expect-cost", "value": "true" } + { + "key": "Content-Type", + "value": "application/json" + }, + { + "key": "x-request-id", + "value": "acct-gemini-nonstreaming" + }, + { + "key": "x-bf-expect-cost", + "value": "true" + } ], "body": { - "mode": "formdata", - "formdata": [ - { "key": "model", "value": "{{transcription_model}}", "type": "text" }, - { "key": "file", "src": "fixtures/sample.mp3", "type": "file" } - ] + "mode": "raw", + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512\n}" }, - "url": { "raw": "{{baseUrl}}/v1/audio/transcriptions", "host": ["{{baseUrl}}"], "path": ["v1", "audio", "transcriptions"] } + "url": { + "raw": "{{baseUrl}}/v1/chat/completions", + "host": [ + "{{baseUrl}}" + ], + "path": [ + "v1", + "chat", + "completions" + ] + } }, "event": [ - { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// 'Other' request type: audio transcription. A failure after the audio was", - "// processed must still record cost. Verified by the dbverify reporter.", - "pm.test('[Costing] transcription request reached the provider', function () {", - " pm.expect(pm.response.code).to.not.equal(404);", - "});" - ] } } + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "// Cost recording is verified against the logs DB by the dbverify reporter", + "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", + "// gateway accepted and routed it.", + "pm.test(`[Accounting] gemini nonstreaming reached the provider`, function () {", + " pm.expect(pm.response.code).to.not.equal(404);", + "});" + ] + } + } ] - } - ] - }, - { - "name": "16. Accounting (cost recording across providers)", - "description": "Criss-cross of providers x {non-streaming, streaming} chat requests that each pin a unique x-request-id and set x-bf-expect-cost. The dbverify reporter asserts the logs DB recorded cost>0 && total_tokens>0 per request, guarding against accounting regressions. Failed-streaming, retry-attempt, cumulative-sum and no-double-bill semantics are covered by Go unit tests (plugins/governance/accounting_test.go).", - "item": [ + }, { - "name": "[Accounting] openai non-streaming records cost", + "name": "[Accounting] gemini streaming records cost", "request": { "method": "POST", "header": [ @@ -8258,7 +36422,7 @@ }, { "key": "x-request-id", - "value": "acct-openai-nonstreaming" + "value": "acct-gemini-streaming" }, { "key": "x-bf-expect-cost", @@ -8267,7 +36431,7 @@ ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512\n}" + "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512,\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", @@ -8290,7 +36454,7 @@ "// Cost recording is verified against the logs DB by the dbverify reporter", "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", "// gateway accepted and routed it.", - "pm.test(`[Accounting] openai nonstreaming reached the provider`, function () {", + "pm.test(`[Accounting] gemini streaming reached the provider`, function () {", " pm.expect(pm.response.code).to.not.equal(404);", "});" ] @@ -8299,7 +36463,7 @@ ] }, { - "name": "[Accounting] openai streaming records cost", + "name": "[Accounting] vertex non-streaming records cost", "request": { "method": "POST", "header": [ @@ -8309,7 +36473,7 @@ }, { "key": "x-request-id", - "value": "acct-openai-streaming" + "value": "acct-vertex-nonstreaming" }, { "key": "x-bf-expect-cost", @@ -8318,7 +36482,7 @@ ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"openai/gpt-4o-mini\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512,\n \"stream\": true\n}" + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", @@ -8341,7 +36505,7 @@ "// Cost recording is verified against the logs DB by the dbverify reporter", "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", "// gateway accepted and routed it.", - "pm.test(`[Accounting] openai streaming reached the provider`, function () {", + "pm.test(`[Accounting] vertex nonstreaming reached the provider`, function () {", " pm.expect(pm.response.code).to.not.equal(404);", "});" ] @@ -8350,7 +36514,7 @@ ] }, { - "name": "[Accounting] anthropic non-streaming records cost", + "name": "[Accounting] vertex streaming records cost", "request": { "method": "POST", "header": [ @@ -8360,7 +36524,7 @@ }, { "key": "x-request-id", - "value": "acct-anthropic-nonstreaming" + "value": "acct-vertex-streaming" }, { "key": "x-bf-expect-cost", @@ -8369,7 +36533,7 @@ ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512\n}" + "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512,\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", @@ -8392,7 +36556,7 @@ "// Cost recording is verified against the logs DB by the dbverify reporter", "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", "// gateway accepted and routed it.", - "pm.test(`[Accounting] anthropic nonstreaming reached the provider`, function () {", + "pm.test(`[Accounting] vertex streaming reached the provider`, function () {", " pm.expect(pm.response.code).to.not.equal(404);", "});" ] @@ -8401,7 +36565,7 @@ ] }, { - "name": "[Accounting] anthropic streaming records cost", + "name": "[Accounting] azure non-streaming records cost", "request": { "method": "POST", "header": [ @@ -8411,7 +36575,7 @@ }, { "key": "x-request-id", - "value": "acct-anthropic-streaming" + "value": "acct-azure-nonstreaming" }, { "key": "x-bf-expect-cost", @@ -8420,7 +36584,7 @@ ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"anthropic/claude-haiku-4-5\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512,\n \"stream\": true\n}" + "raw": "{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", @@ -8443,7 +36607,7 @@ "// Cost recording is verified against the logs DB by the dbverify reporter", "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", "// gateway accepted and routed it.", - "pm.test(`[Accounting] anthropic streaming reached the provider`, function () {", + "pm.test(`[Accounting] azure nonstreaming reached the provider`, function () {", " pm.expect(pm.response.code).to.not.equal(404);", "});" ] @@ -8452,7 +36616,7 @@ ] }, { - "name": "[Accounting] bedrock non-streaming records cost", + "name": "[Accounting] azure streaming records cost", "request": { "method": "POST", "header": [ @@ -8462,7 +36626,7 @@ }, { "key": "x-request-id", - "value": "acct-bedrock-nonstreaming" + "value": "acct-azure-streaming" }, { "key": "x-bf-expect-cost", @@ -8471,7 +36635,7 @@ ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512\n}" + "raw": "{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512,\n \"stream\": true\n}" }, "url": { "raw": "{{baseUrl}}/v1/chat/completions", @@ -8494,67 +36658,91 @@ "// Cost recording is verified against the logs DB by the dbverify reporter", "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", "// gateway accepted and routed it.", - "pm.test(`[Accounting] bedrock nonstreaming reached the provider`, function () {", + "pm.test(`[Accounting] azure streaming reached the provider`, function () {", " pm.expect(pm.response.code).to.not.equal(404);", "});" ] } } ] - }, + } + ] + }, + { + "name": "17. Provider Egress Streaming/Truncation Guards (#4923 / #4932 / #4680)", + "item": [ { - "name": "[Accounting] bedrock streaming records cost", + "name": "Bedrock converse-stream (forced tool) closes the tool_use block - #4923", + "event": [ + { + "listen": "test", + "script": { + "type": "text/javascript", + "exec": [ + "if (pm.response.code >= 400) { return; }", + "pm.test('Bedrock forced-tool converse-stream closes content blocks - #4923', function () {", + " var body = pm.response.text() || '';", + " pm.expect(body, 'expected a toolUse block in the forced-tool stream').to.include('toolUse');", + " pm.expect(body, 'expected a contentBlockStop event closing each content block').to.include('contentBlockStop');", + " pm.expect(body, 'expected a messageStop event terminating the stream').to.include('messageStop');", + " pm.expect(body.indexOf('contentBlockStop'), 'contentBlockStop must precede the terminal messageStop').to.be.below(body.lastIndexOf('messageStop'));", + "});" + ] + } + } + ], "request": { "method": "POST", "header": [ { "key": "Content-Type", "value": "application/json" - }, - { - "key": "x-request-id", - "value": "acct-bedrock-streaming" - }, - { - "key": "x-bf-expect-cost", - "value": "true" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"bedrock/us.amazon.nova-lite-v1:0\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512,\n \"stream\": true\n}" + "raw": "{\"messages\": [{\"role\": \"user\", \"content\": [{\"text\": \"What is the weather in San Francisco right now? Use the get_weather tool.\"}]}], \"inferenceConfig\": {\"maxTokens\": 512}, \"toolConfig\": {\"tools\": [{\"toolSpec\": {\"name\": \"get_weather\", \"description\": \"Get the current weather for a city\", \"inputSchema\": {\"json\": {\"type\": \"object\", \"properties\": {\"city\": {\"type\": \"string\"}}, \"required\": [\"city\"]}}}}], \"toolChoice\": {\"any\": {}}}}" }, "url": { - "raw": "{{baseUrl}}/v1/chat/completions", + "raw": "{{baseUrl}}/bedrock/model/{{bedrockModel}}/converse-stream", "host": [ "{{baseUrl}}" ], "path": [ - "v1", - "chat", - "completions" + "bedrock", + "model", + "{{bedrockModel}}", + "converse-stream" ] } - }, + } + }, + { + "name": "Anthropic normalized web_fetch streaming has contiguous content_block indices - #4932", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// Cost recording is verified against the logs DB by the dbverify reporter", - "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", - "// gateway accepted and routed it.", - "pm.test(`[Accounting] bedrock streaming reached the provider`, function () {", - " pm.expect(pm.response.code).to.not.equal(404);", + "if (pm.response.code >= 400) { return; }", + "// If newman forwards the custom User-Agent, Bifrost takes the Claude Code passthrough", + "// path; if it strips it, this still validates the all-normalized path. Either way the", + "// content_block_start indices must be contiguous from 0 (#4890 / #4932).", + "pm.test('Anthropic normalized web_fetch stream: content_block indices contiguous from 0 - #4932', function () {", + " var raw = pm.response.text() || '';", + " pm.expect(raw, 'stream did not end with message_stop').to.include('message_stop');", + " pm.expect(raw.indexOf('Content block not found'), 'stream reported a missing content block').to.equal(-1);", + " var idxs = [], m, re = /\"content_block_start\"[^}]*?\"index\"\\s*:\\s*(\\d+)/g;", + " while ((m = re.exec(raw)) !== null) { idxs.push(Number(m[1])); }", + " pm.expect(idxs.length, 'no content_block_start events found in stream').to.be.above(0);", + " var srt = idxs.slice().sort(function (a, b) { return a - b; });", + " for (var i = 0; i < srt.length; i++) { pm.expect(srt[i], 'content_block_start indices not contiguous from 0: ' + JSON.stringify(idxs)).to.equal(i); }", "});" ] } } - ] - }, - { - "name": "[Accounting] gemini non-streaming records cost", + ], "request": { "method": "POST", "header": [ @@ -8563,151 +36751,147 @@ "value": "application/json" }, { - "key": "x-request-id", - "value": "acct-gemini-nonstreaming" + "key": "x-api-key", + "value": "{{anthropicKey}}" }, { - "key": "x-bf-expect-cost", - "value": "true" + "key": "anthropic-version", + "value": "2023-06-01" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512\n}" + "raw": "{\"model\": \"claude-opus-4-7\", \"max_tokens\": 1024, \"messages\": [{\"role\": \"user\", \"content\": \"Fetch https://example.com and give me a one-sentence summary.\"}], \"tools\": [{\"type\": \"web_fetch_20250910\", \"name\": \"web_fetch\", \"max_uses\": 2}], \"stream\": true}" }, "url": { - "raw": "{{baseUrl}}/v1/chat/completions", + "raw": "{{baseUrl}}/anthropic/v1/messages", "host": [ "{{baseUrl}}" ], "path": [ + "anthropic", "v1", - "chat", - "completions" + "messages" ] } - }, + } + }, + { + "name": "Bedrock Responses max_output_tokens truncation signals incomplete (non-streaming) - #4680", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// Cost recording is verified against the logs DB by the dbverify reporter", - "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", - "// gateway accepted and routed it.", - "pm.test(`[Accounting] gemini nonstreaming reached the provider`, function () {", - " pm.expect(pm.response.code).to.not.equal(404);", + "if (pm.response.code >= 400) { return; }", + "pm.test('Bedrock Responses truncation -> status=incomplete, reason=max_output_tokens - #4680', function () {", + " var j = pm.response.json();", + " pm.expect(j.status, 'expected status=incomplete on a truncated response').to.equal('incomplete');", + " pm.expect(j.incomplete_details && j.incomplete_details.reason, 'expected incomplete_details.reason=max_output_tokens').to.equal('max_output_tokens');", "});" ] } } - ] - }, - { - "name": "[Accounting] gemini streaming records cost", + ], "request": { "method": "POST", "header": [ { "key": "Content-Type", "value": "application/json" - }, - { - "key": "x-request-id", - "value": "acct-gemini-streaming" - }, - { - "key": "x-bf-expect-cost", - "value": "true" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"gemini/gemini-2.5-flash\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512,\n \"stream\": true\n}" + "raw": "{\"model\": \"bedrock/us.amazon.nova-lite-v1:0\", \"input\": \"Write a detailed 500-word essay about the history and design philosophy of the Unix operating system.\", \"max_output_tokens\": 16}" }, "url": { - "raw": "{{baseUrl}}/v1/chat/completions", + "raw": "{{baseUrl}}/v1/responses", "host": [ "{{baseUrl}}" ], "path": [ "v1", - "chat", - "completions" + "responses" ] } - }, + } + }, + { + "name": "Bedrock Responses max_output_tokens truncation emits response.incomplete (streaming) - #4680", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// Cost recording is verified against the logs DB by the dbverify reporter", - "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", - "// gateway accepted and routed it.", - "pm.test(`[Accounting] gemini streaming reached the provider`, function () {", - " pm.expect(pm.response.code).to.not.equal(404);", + "if (pm.response.code >= 400) { return; }", + "pm.test('Bedrock Responses streaming truncation emits response.incomplete (not response.completed) - #4680', function () {", + " var raw = pm.response.text() || '';", + " pm.expect(raw, 'expected a response.incomplete terminal event').to.include('response.incomplete');", + " pm.expect(raw, 'expected the incomplete reason max_output_tokens').to.include('max_output_tokens');", + " pm.expect(raw.indexOf('response.completed'), 'a truncated stream must not emit response.completed').to.equal(-1);", "});" ] } } - ] - }, - { - "name": "[Accounting] vertex non-streaming records cost", + ], "request": { "method": "POST", "header": [ { "key": "Content-Type", "value": "application/json" - }, - { - "key": "x-request-id", - "value": "acct-vertex-nonstreaming" - }, - { - "key": "x-bf-expect-cost", - "value": "true" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512\n}" + "raw": "{\"model\": \"bedrock/us.amazon.nova-lite-v1:0\", \"input\": \"Write a detailed 500-word essay about the history and design philosophy of the Unix operating system.\", \"max_output_tokens\": 16, \"stream\": true}" }, "url": { - "raw": "{{baseUrl}}/v1/chat/completions", + "raw": "{{baseUrl}}/v1/responses", "host": [ "{{baseUrl}}" ], "path": [ "v1", - "chat", - "completions" + "responses" ] } - }, + } + } + ] + }, + { + "name": "18. Claude Code Passthrough - server-tool streaming index contiguity (#4890)", + "item": [ + { + "name": "Claude Code passthrough: web_search streaming has contiguous content_block indices - #4890", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// Cost recording is verified against the logs DB by the dbverify reporter", - "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", - "// gateway accepted and routed it.", - "pm.test(`[Accounting] vertex nonstreaming reached the provider`, function () {", - " pm.expect(pm.response.code).to.not.equal(404);", + "if (pm.response.code >= 400) { return; }", + "// If newman forwards the custom User-Agent, Bifrost takes the Claude Code passthrough", + "// path; if it strips it, this still validates the all-normalized path. Either way the", + "// content_block_start indices must be contiguous from 0 (#4890 / #4932).", + "pm.test('Claude Code passthrough web_search stream: content_block indices contiguous from 0 - #4890', function () {", + " var raw = pm.response.text() || '';", + " pm.expect(raw, 'stream did not end with message_stop').to.include('message_stop');", + " pm.expect(raw.indexOf('Content block not found'), 'stream reported a missing content block').to.equal(-1);", + " var idxs = [], m, re = /\"content_block_start\"[^}]*?\"index\"\\s*:\\s*(\\d+)/g;", + " while ((m = re.exec(raw)) !== null) { idxs.push(Number(m[1])); }", + " pm.expect(idxs.length, 'no content_block_start events found in stream').to.be.above(0);", + " var srt = idxs.slice().sort(function (a, b) { return a - b; });", + " for (var i = 0; i < srt.length; i++) { pm.expect(srt[i], 'content_block_start indices not contiguous from 0: ' + JSON.stringify(idxs)).to.equal(i); }", "});" ] } } - ] - }, - { - "name": "[Accounting] vertex streaming records cost", + ], "request": { "method": "POST", "header": [ @@ -8716,49 +36900,61 @@ "value": "application/json" }, { - "key": "x-request-id", - "value": "acct-vertex-streaming" + "key": "x-api-key", + "value": "{{anthropicKey}}" }, { - "key": "x-bf-expect-cost", - "value": "true" + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "User-Agent", + "value": "claude-cli/1.0" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"vertex/gemini-2.5-flash\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512,\n \"stream\": true\n}" + "raw": "{\"model\": \"claude-opus-4-7\", \"max_tokens\": 1024, \"messages\": [{\"role\": \"user\", \"content\": \"What's the weather in New York City right now?\"}], \"tools\": [{\"type\": \"web_search_20250305\", \"name\": \"web_search\", \"max_uses\": 3}], \"stream\": true}" }, "url": { - "raw": "{{baseUrl}}/v1/chat/completions", + "raw": "{{baseUrl}}/anthropic/v1/messages", "host": [ "{{baseUrl}}" ], "path": [ + "anthropic", "v1", - "chat", - "completions" + "messages" ] } - }, + } + }, + { + "name": "Claude Code passthrough: zero-result web_search keeps indices in lockstep - #4890 (f2c)", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// Cost recording is verified against the logs DB by the dbverify reporter", - "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", - "// gateway accepted and routed it.", - "pm.test(`[Accounting] vertex streaming reached the provider`, function () {", - " pm.expect(pm.response.code).to.not.equal(404);", + "if (pm.response.code >= 400) { return; }", + "// If newman forwards the custom User-Agent, Bifrost takes the Claude Code passthrough", + "// path; if it strips it, this still validates the all-normalized path. Either way the", + "// content_block_start indices must be contiguous from 0 (#4890 / #4932).", + "pm.test('Claude Code passthrough zero-result web_search: content_block indices still contiguous - #4890', function () {", + " var raw = pm.response.text() || '';", + " pm.expect(raw, 'stream did not end with message_stop').to.include('message_stop');", + " pm.expect(raw.indexOf('Content block not found'), 'stream reported a missing content block').to.equal(-1);", + " var idxs = [], m, re = /\"content_block_start\"[^}]*?\"index\"\\s*:\\s*(\\d+)/g;", + " while ((m = re.exec(raw)) !== null) { idxs.push(Number(m[1])); }", + " pm.expect(idxs.length, 'no content_block_start events found in stream').to.be.above(0);", + " var srt = idxs.slice().sort(function (a, b) { return a - b; });", + " for (var i = 0; i < srt.length; i++) { pm.expect(srt[i], 'content_block_start indices not contiguous from 0: ' + JSON.stringify(idxs)).to.equal(i); }", "});" ] } } - ] - }, - { - "name": "[Accounting] azure non-streaming records cost", + ], "request": { "method": "POST", "header": [ @@ -8767,49 +36963,61 @@ "value": "application/json" }, { - "key": "x-request-id", - "value": "acct-azure-nonstreaming" + "key": "x-api-key", + "value": "{{anthropicKey}}" }, { - "key": "x-bf-expect-cost", - "value": "true" + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "User-Agent", + "value": "claude-cli/1.0" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512\n}" + "raw": "{\"model\": \"claude-opus-4-7\", \"max_tokens\": 1024, \"messages\": [{\"role\": \"user\", \"content\": \"Search the web for the exact phrase \\\"zxqw9v7k3mfp2qh8 nonexistent gibberish token 4823\\\" and tell me exactly what search results come back.\"}], \"tools\": [{\"type\": \"web_search_20250305\", \"name\": \"web_search\", \"max_uses\": 3}], \"stream\": true}" }, "url": { - "raw": "{{baseUrl}}/v1/chat/completions", + "raw": "{{baseUrl}}/anthropic/v1/messages", "host": [ "{{baseUrl}}" ], "path": [ + "anthropic", "v1", - "chat", - "completions" + "messages" ] } - }, + } + }, + { + "name": "Claude Code passthrough: web_fetch streaming has contiguous content_block indices - #4890 (5464)", "event": [ { "listen": "test", "script": { "type": "text/javascript", "exec": [ - "// Cost recording is verified against the logs DB by the dbverify reporter", - "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", - "// gateway accepted and routed it.", - "pm.test(`[Accounting] azure nonstreaming reached the provider`, function () {", - " pm.expect(pm.response.code).to.not.equal(404);", + "if (pm.response.code >= 400) { return; }", + "// If newman forwards the custom User-Agent, Bifrost takes the Claude Code passthrough", + "// path; if it strips it, this still validates the all-normalized path. Either way the", + "// content_block_start indices must be contiguous from 0 (#4890 / #4932).", + "pm.test('Claude Code passthrough web_fetch stream: content_block indices contiguous from 0 - #4890', function () {", + " var raw = pm.response.text() || '';", + " pm.expect(raw, 'stream did not end with message_stop').to.include('message_stop');", + " pm.expect(raw.indexOf('Content block not found'), 'stream reported a missing content block').to.equal(-1);", + " var idxs = [], m, re = /\"content_block_start\"[^}]*?\"index\"\\s*:\\s*(\\d+)/g;", + " while ((m = re.exec(raw)) !== null) { idxs.push(Number(m[1])); }", + " pm.expect(idxs.length, 'no content_block_start events found in stream').to.be.above(0);", + " var srt = idxs.slice().sort(function (a, b) { return a - b; });", + " for (var i = 0; i < srt.length; i++) { pm.expect(srt[i], 'content_block_start indices not contiguous from 0: ' + JSON.stringify(idxs)).to.equal(i); }", "});" ] } } - ] - }, - { - "name": "[Accounting] azure streaming records cost", + ], "request": { "method": "POST", "header": [ @@ -8818,48 +37026,36 @@ "value": "application/json" }, { - "key": "x-request-id", - "value": "acct-azure-streaming" + "key": "x-api-key", + "value": "{{anthropicKey}}" }, { - "key": "x-bf-expect-cost", - "value": "true" + "key": "anthropic-version", + "value": "2023-06-01" + }, + { + "key": "User-Agent", + "value": "claude-cli/1.0" } ], "body": { "mode": "raw", - "raw": "{\n \"model\": \"azure/{{azureDeployment}}\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Reply with a short sentence about the number seven.\"\n }\n ],\n \"max_tokens\": 512,\n \"stream\": true\n}" + "raw": "{\"model\": \"claude-opus-4-7\", \"max_tokens\": 1024, \"messages\": [{\"role\": \"user\", \"content\": \"Fetch https://example.com and summarize it in one sentence.\"}], \"tools\": [{\"type\": \"web_fetch_20250910\", \"name\": \"web_fetch\", \"max_uses\": 2}], \"stream\": true}" }, "url": { - "raw": "{{baseUrl}}/v1/chat/completions", + "raw": "{{baseUrl}}/anthropic/v1/messages", "host": [ "{{baseUrl}}" ], "path": [ + "anthropic", "v1", - "chat", - "completions" + "messages" ] } - }, - "event": [ - { - "listen": "test", - "script": { - "type": "text/javascript", - "exec": [ - "// Cost recording is verified against the logs DB by the dbverify reporter", - "// (cost>0 && total_tokens>0 for this x-request-id). This request asserts the", - "// gateway accepted and routed it.", - "pm.test(`[Accounting] azure streaming reached the provider`, function () {", - " pm.expect(pm.response.code).to.not.equal(404);", - "});" - ] - } - } - ] + } } ] } ] -} +} \ No newline at end of file diff --git a/tests/e2e/api/provider-capabilities.json b/tests/e2e/api/provider-capabilities.json index 0eb25031236..c9bc21b0355 100644 --- a/tests/e2e/api/provider-capabilities.json +++ b/tests/e2e/api/provider-capabilities.json @@ -177,6 +177,94 @@ "video_remix": false, "rerank": true }, + "bedrock_mantle": { + "chat_completions": true, + "chat_completions_with_tools": true, + "text_completion": false, + "responses": true, + "responses_with_tools": true, + "count_tokens": false, + "embedding": false, + "speech": false, + "transcription": false, + "list_models": true, + "image_generation": false, + "image_variation": false, + "image_edit": false, + "batch_create": false, + "batch_list": false, + "batch_retrieve": false, + "batch_cancel": false, + "batch_results": false, + "file_batch_input": false, + "batch_create_file": false, + "file_upload": false, + "file_list": false, + "file_retrieve": false, + "file_delete": false, + "file_content": false, + "container_create": false, + "container_list": false, + "container_retrieve": false, + "container_delete": false, + "container_file_create": false, + "container_file_create_reference": false, + "container_file_list": false, + "container_file_retrieve": false, + "container_file_content": false, + "container_file_delete": false, + "video_generation": false, + "video_retrieve": false, + "video_download": false, + "video_delete": false, + "video_list": false, + "video_remix": false, + "rerank": false + }, + "deepseek": { + "chat_completions": true, + "chat_completions_with_tools": true, + "text_completion": true, + "responses": false, + "responses_with_tools": false, + "count_tokens": false, + "embedding": false, + "speech": false, + "transcription": false, + "list_models": true, + "image_generation": false, + "image_variation": false, + "image_edit": false, + "batch_create": false, + "batch_list": false, + "batch_retrieve": false, + "batch_cancel": false, + "batch_results": false, + "file_batch_input": false, + "batch_create_file": false, + "file_upload": false, + "file_list": false, + "file_retrieve": false, + "file_delete": false, + "file_content": false, + "container_create": false, + "container_list": false, + "container_retrieve": false, + "container_delete": false, + "container_file_create": false, + "container_file_create_reference": false, + "container_file_list": false, + "container_file_retrieve": false, + "container_file_content": false, + "container_file_delete": false, + "video_generation": false, + "video_retrieve": false, + "video_download": false, + "video_delete": false, + "video_list": false, + "video_remix": false, + "rerank": false + }, "cerebras": { "chat_completions": true, "chat_completions_with_tools": true, @@ -794,4 +882,4 @@ "rerank": false } } -} \ No newline at end of file +} diff --git a/tests/e2e/api/provider_config/bifrost-v1-bedrock-mantle.postman_environment.json b/tests/e2e/api/provider_config/bifrost-v1-bedrock-mantle.postman_environment.json new file mode 100644 index 00000000000..83d924f129e --- /dev/null +++ b/tests/e2e/api/provider_config/bifrost-v1-bedrock-mantle.postman_environment.json @@ -0,0 +1,120 @@ +{ + "id": "bifrost-v1-env-bedrock-mantle", + "name": "Bifrost V1 – bedrock_mantle", + "values": [ + { + "key": "base_url", + "value": "http://localhost:8080", + "type": "default", + "enabled": true + }, + { + "key": "provider", + "value": "bedrock_mantle", + "type": "default", + "enabled": true + }, + { + "key": "model", + "value": "anthropic.claude-opus-4-8", + "type": "default", + "enabled": true + }, + { + "key": "chat_model", + "value": "anthropic.claude-opus-4-8", + "type": "default", + "enabled": true + }, + { + "key": "responses_model", + "value": "anthropic.claude-opus-4-8", + "type": "default", + "enabled": true + }, + { + "key": "text_completion_model", + "value": "unsupported_text_completion_model", + "type": "default", + "enabled": true + }, + { + "key": "embedding_model", + "value": "unsupported_embedding_model", + "type": "default", + "enabled": true + }, + { + "key": "speech_model", + "value": "unsupported_speech_model", + "type": "default", + "enabled": true + }, + { + "key": "transcription_model", + "value": "unsupported_transcription_model", + "type": "default", + "enabled": true + }, + { + "key": "image_model", + "value": "unsupported_image_model", + "type": "default", + "enabled": true + }, + { + "key": "image_variation_model", + "value": "unsupported_image_variation_model", + "type": "default", + "enabled": true + }, + { + "key": "bedrock_mantle_api_key", + "value": "", + "type": "secret", + "enabled": true + }, + { + "key": "bedrock_mantle_access_key", + "value": "", + "type": "secret", + "enabled": true + }, + { + "key": "bedrock_mantle_secret_key", + "value": "", + "type": "secret", + "enabled": true + }, + { + "key": "bedrock_mantle_region", + "value": "us-east-1", + "type": "default", + "enabled": true + }, + { + "key": "bedrock_mantle_session_token", + "value": "", + "type": "secret", + "enabled": true + }, + { + "key": "batch_id", + "value": "batch_123", + "type": "default", + "enabled": true + }, + { + "key": "file_id", + "value": "file_123", + "type": "default", + "enabled": true + }, + { + "key": "container_id", + "value": "container_123", + "type": "default", + "enabled": true + } + ] +} diff --git a/tests/e2e/api/runners/filter-collection.mjs b/tests/e2e/api/runners/filter-collection.mjs index a5132c04f20..aa5bdb2f825 100644 --- a/tests/e2e/api/runners/filter-collection.mjs +++ b/tests/e2e/api/runners/filter-collection.mjs @@ -48,6 +48,7 @@ const PROVIDER_KEYWORDS = { openai: ["openai", "/openai", "gpt-", "o3", "o1"], anthropic: ["anthropic", "claude-"], bedrock: ["bedrock", "/bedrock"], + bedrock_mantle: ["bedrock_mantle", "bedrock-mantle"], gemini: ["gemini", "/genai", "googlesearch"], vertex: ["vertex", "/genai/v1beta/models/{{vertexModel}}"], azure: ["azure", "deployments"], @@ -109,6 +110,11 @@ const itemMatchesProvider = (item, ancestorNames) => { const isOpenRouter = haystack.includes("openrouter"); if (PROVIDER === "openrouter") return isOpenRouter; if (isOpenRouter) return false; + // Bedrock Mantle rows (model "bedrock_mantle/...") contain the substring "bedrock", so they'd + // otherwise be claimed by the bedrock partition too. Route them exclusively to bedrock_mantle. + const isMantle = haystack.includes("bedrock_mantle") || haystack.includes("bedrock-mantle"); + if (PROVIDER === "bedrock_mantle") return isMantle; + if (isMantle) return false; return keywords.some((k) => haystack.includes(k)); }; diff --git a/tests/e2e/api/runners/harness-monitor.mjs b/tests/e2e/api/runners/harness-monitor.mjs index 118c9a9fd62..742958a0fd6 100644 --- a/tests/e2e/api/runners/harness-monitor.mjs +++ b/tests/e2e/api/runners/harness-monitor.mjs @@ -178,7 +178,7 @@ function readNewBytes() { // ----- Parsing ---------------------------------------------------------------- -const RE_PREFIX = /^\[([a-z]+)\]\s?(.*)$/; +const RE_PREFIX = /^\[([a-z_]+)\]\s?(.*)$/; const RE_FOLDER = /^❏\s+(.+?)\s*$/; const RE_REQUEST = /^↳\s+(.+?)\s*$/; const RE_REQUEST_DONE = /\[\s*\d+(?:\s+[A-Za-z]+)?,\s*[\d.]+\s*[kMG]?B,\s*[\d.]+\s*m?s\s*\]/; diff --git a/tests/e2e/api/runners/individual/run-newman-mcp-auth-tests.sh b/tests/e2e/api/runners/individual/run-newman-mcp-auth-tests.sh new file mode 100755 index 00000000000..f98f4cbf27d --- /dev/null +++ b/tests/e2e/api/runners/individual/run-newman-mcp-auth-tests.sh @@ -0,0 +1,290 @@ +#!/bin/bash + +# Bifrost MCP Auth Newman Test Runner +# +# Boots a fresh Bifrost server (sqlite, pre-seeded virtual keys + an upstream MCP +# client) once per inbound /mcp authentication mode: +# - client.mcp_server_auth_mode = headers (default; header credentials only, discovery off) +# - client.mcp_server_auth_mode = both (header credentials AND issued JWTs, discovery on) +# - client.mcp_server_auth_mode = oauth (issued JWTs only, header credentials rejected) +# and runs collections/bifrost-v1-mcp-auth.postman_collection.json against each. +# +# The collection's test scripts branch on the auth_mode env var, so a single +# collection encodes the full accept/reject matrix. The central guarantee: in +# headers mode every existing virtual-key path connects exactly as before and the +# OAuth surface is invisible (discovery 404s); enabling both only ADDS JWT +# acceptance without changing any header-credential outcome. +# +# An upstream MCP server (examples/mcps/http-no-ping-server on port 3001) is built +# and started by this runner so /mcp exposes real tools. It is pre-seeded as an +# MCP client in each boot config, so it is connected before the server reports +# ready — no post-boot registration race. +# +# Requires a built bifrost-http binary; this runner boots its own servers (each +# mode needs a different boot config, so the shared e2e server is not reused). + +set -e + +SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" +API_DIR="$(cd "$SCRIPT_DIR/../.." && pwd)" +REPO_ROOT="$(cd "$API_DIR/../../.." && pwd)" +cd "$API_DIR" + +COLLECTION="collections/bifrost-v1-mcp-auth.postman_collection.json" +REPORT_DIR="newman-reports/mcp-auth" +VK_VALUE="sk-bf-mcp-test-key" +VK_INACTIVE_VALUE="sk-bf-mcp-inactive-key" +MCP_UPSTREAM_DIR="$REPO_ROOT/examples/mcps/http-no-ping-server" +MCP_UPSTREAM_PORT="3001" + +GREEN='\033[0;32m' +YELLOW='\033[1;33m' +RED='\033[0;31m' +NC='\033[0m' + +BIFROST_BINARY="" +PORT="8090" +REPORTERS="cli" +VERBOSE="" +BAIL="" + +while [[ $# -gt 0 ]]; do + case "$1" in + --binary) BIFROST_BINARY="$2"; shift 2 ;; + --port) PORT="$2"; shift 2 ;; + --mcp-port) MCP_UPSTREAM_PORT="$2"; shift 2 ;; + --html) REPORTERS="${REPORTERS},html"; shift ;; + --json) REPORTERS="${REPORTERS},json"; shift ;; + --verbose) VERBOSE="--verbose"; shift ;; + --bail) BAIL="--bail"; shift ;; + --help) + echo "Usage: $0 --binary [OPTIONS]" + echo "" + echo "Options:" + echo " --binary Path to a built bifrost-http binary (required)" + echo " --port Port to boot each server on (default: 8090)" + echo " --mcp-port Port for the upstream MCP test server (default: 3001)" + echo " --html Generate HTML report" + echo " --json Generate JSON report" + echo " --verbose Show detailed Newman output" + echo " --bail Stop on first failure" + echo " --help Show this help message" + exit 0 ;; + *) echo -e "${RED}Unknown option: $1${NC}"; exit 1 ;; + esac +done + +echo -e "${GREEN}==============================================${NC}" +echo -e "${GREEN}Bifrost MCP Auth Test Runner${NC}" +echo -e "${GREEN}==============================================${NC}" +echo "" + +if ! command -v newman &>/dev/null; then + echo -e "${RED}Error: Newman is not installed${NC}" + echo "Install it with: npm install -g newman" + exit 1 +fi +if [ -z "$BIFROST_BINARY" ] || [ ! -x "$BIFROST_BINARY" ]; then + echo -e "${RED}Error: --binary must point to an executable bifrost-http binary${NC}" + exit 1 +fi +if [ ! -f "$COLLECTION" ]; then + echo -e "${RED}Error: Collection file not found: $COLLECTION${NC}" + exit 1 +fi + +mkdir -p "$REPORT_DIR" + +# Tracks the running servers + temp dirs so the trap can always clean up. +CURRENT_PID="" +CURRENT_DIR="" +MCP_UPSTREAM_PID="" +MCP_UPSTREAM_TMP="" +OVERALL_EXIT=0 + +cleanup() { + if [ -n "$CURRENT_PID" ] && kill -0 "$CURRENT_PID" 2>/dev/null; then + kill "$CURRENT_PID" 2>/dev/null || true + wait "$CURRENT_PID" 2>/dev/null || true + fi + [ -n "$CURRENT_DIR" ] && rm -rf "$CURRENT_DIR" + [ -n "$MCP_UPSTREAM_TMP" ] && rm -rf "$MCP_UPSTREAM_TMP" + if [ -n "$MCP_UPSTREAM_PID" ] && kill -0 "$MCP_UPSTREAM_PID" 2>/dev/null; then + kill "$MCP_UPSTREAM_PID" 2>/dev/null || true + wait "$MCP_UPSTREAM_PID" 2>/dev/null || true + fi +} +trap cleanup EXIT + +# Build and start the upstream MCP server so /mcp has real tools to expose. +start_upstream_mcp() { + if [ ! -d "$MCP_UPSTREAM_DIR" ]; then + echo -e "${RED}Error: upstream MCP server source not found: $MCP_UPSTREAM_DIR${NC}" + exit 1 + fi + # Fail fast with a clear message if the chosen port is already taken, rather + # than letting the server fail to bind and surfacing as a readiness timeout. + if (command -v nc &>/dev/null && nc -z 127.0.0.1 "$MCP_UPSTREAM_PORT" 2>/dev/null) \ + || (echo >/dev/tcp/127.0.0.1/"$MCP_UPSTREAM_PORT") 2>/dev/null; then + echo -e "${RED}Error: port $MCP_UPSTREAM_PORT is already in use; pass --mcp-port to use another${NC}" + exit 1 + fi + echo -e "${YELLOW}Building upstream MCP server...${NC}" + # Build into a temp dir (cleaned up on exit) so no binary is left in the + # example's source tree. GOWORK=off: the example has its own module and is + # not part of the repo workspace, so building it in-workspace would fail. + MCP_UPSTREAM_TMP="$(mktemp -d)" + ( cd "$MCP_UPSTREAM_DIR" && GOWORK=off go build -o "$MCP_UPSTREAM_TMP/http-no-ping-server" . ) || { + echo -e "${RED}Error: failed to build upstream MCP server${NC}"; exit 1; } + MCP_SERVER_PORT="$MCP_UPSTREAM_PORT" "$MCP_UPSTREAM_TMP/http-no-ping-server" > "$API_DIR/$REPORT_DIR/upstream-mcp.log" 2>&1 & + MCP_UPSTREAM_PID=$! + local waited=0 + while [ $waited -lt 20 ]; do + if (command -v nc &>/dev/null && nc -z 127.0.0.1 "$MCP_UPSTREAM_PORT" 2>/dev/null) \ + || (echo >/dev/tcp/127.0.0.1/"$MCP_UPSTREAM_PORT") 2>/dev/null; then + echo -e "${GREEN}Upstream MCP server ready on :$MCP_UPSTREAM_PORT${NC}" + return + fi + if ! kill -0 "$MCP_UPSTREAM_PID" 2>/dev/null; then + echo -e "${RED}Upstream MCP server exited during startup${NC}" + cat "$API_DIR/$REPORT_DIR/upstream-mcp.log" + exit 1 + fi + sleep 0.5; waited=$((waited + 1)) + done + echo -e "${RED}Upstream MCP server did not become ready${NC}"; exit 1 +} + +# write_config +write_config() { + local dir="$1" mode="$2" + local oauth_block="" + # Discovery + JWT issuance only apply in both/oauth. A stable issuer_url keeps + # discovery docs and minted-token claims deterministic across the collection. + if [ "$mode" != "headers" ]; then + oauth_block="\"oauth2_server_config\": { \"issuer_url\": \"http://localhost:$PORT\", \"auth_code_ttl\": 600, \"access_token_ttl\": 600 }," + fi + cat > "$dir/config.json" < +run_mode() { + local mode="$1" + echo -e "${GREEN}----------------------------------------------${NC}" + echo -e "${GREEN}MCP server auth mode: ${YELLOW}${mode}${NC}" + echo -e "${GREEN}----------------------------------------------${NC}" + + CURRENT_DIR="$(mktemp -d)" + write_config "$CURRENT_DIR" "$mode" + local server_log="$CURRENT_DIR/server.log" + + "$BIFROST_BINARY" --app-dir "$CURRENT_DIR" --port "$PORT" --log-level info > "$server_log" 2>&1 & + CURRENT_PID=$! + + local elapsed=0 + while [ $elapsed -lt 60 ]; do + grep -q "successfully started bifrost" "$server_log" 2>/dev/null && break + if ! kill -0 "$CURRENT_PID" 2>/dev/null; then + echo -e "${RED} Server exited before becoming ready${NC}"; cat "$server_log" + OVERALL_EXIT=1; CURRENT_PID=""; rm -rf "$CURRENT_DIR"; CURRENT_DIR=""; return + fi + sleep 1; elapsed=$((elapsed + 1)) + done + if [ $elapsed -ge 60 ]; then + echo -e "${RED} Server did not start within 60s${NC}"; cat "$server_log"; OVERALL_EXIT=1 + else + local report_prefix="${REPORT_DIR}/${mode}" + local cmd=(newman run "$COLLECTION" + --env-var "base_url=http://localhost:$PORT" + --env-var "auth_mode=$mode" + --env-var "vk_value=$VK_VALUE" + --env-var "vk_inactive_value=$VK_INACTIVE_VALUE" + --env-var "mcp_issuer=http://localhost:$PORT" + --timeout-script 60000 --timeout 120000 + --ignore-redirects + -r "$REPORTERS") + [[ "$REPORTERS" == *"html"* ]] && cmd+=(--reporter-html-export "${report_prefix}.html") + [[ "$REPORTERS" == *"json"* ]] && cmd+=(--reporter-json-export "${report_prefix}.json") + [ -n "$VERBOSE" ] && cmd+=("$VERBOSE") + [ -n "$BAIL" ] && cmd+=("$BAIL") + + set +e + "${cmd[@]}" + local code=$? + set -e + [ $code -ne 0 ] && OVERALL_EXIT=1 + fi + + if [ -n "$CURRENT_PID" ] && kill -0 "$CURRENT_PID" 2>/dev/null; then + kill "$CURRENT_PID" 2>/dev/null || true + wait "$CURRENT_PID" 2>/dev/null || true + fi + CURRENT_PID="" + rm -rf "$CURRENT_DIR"; CURRENT_DIR="" + echo "" +} + +start_upstream_mcp +run_mode "headers" +run_mode "both" +run_mode "oauth" + +echo "" +if [ $OVERALL_EXIT -eq 0 ]; then + echo -e "${GREEN}✓ All MCP auth modes passed!${NC}" +else + echo -e "${RED}✗ Some MCP auth mode checks failed${NC}" +fi +if [[ "$REPORTERS" == *"html"* ]] || [[ "$REPORTERS" == *"json"* ]]; then + echo "" + echo -e "Reports saved to: ${YELLOW}$REPORT_DIR${NC}" + ls -lh "$REPORT_DIR" 2>/dev/null | tail -n +2 +fi + +exit $OVERALL_EXIT diff --git a/tests/e2e/api/runners/individual/run-newman-vk-expiry-tests.sh b/tests/e2e/api/runners/individual/run-newman-vk-expiry-tests.sh new file mode 100755 index 00000000000..e4a4a3c59a7 --- /dev/null +++ b/tests/e2e/api/runners/individual/run-newman-vk-expiry-tests.sh @@ -0,0 +1,198 @@ +#!/bin/bash + +# Bifrost V1 Virtual Key Expiry Newman Test Runner +# Runs VK expiry tests: create/update validation, runtime expiry enforcement, and cleanup. + +set -e + +SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" +API_DIR="$(cd "$SCRIPT_DIR/../.." && pwd)" +cd "$API_DIR" + +# Configuration +COLLECTION="collections/bifrost-v1-vk-expiry.postman_collection.json" +REPORT_DIR="newman-reports/vk-expiry" +PROVIDER_CONFIG_DIR="provider_config" +PROVIDER_CAPABILITIES_JSON="provider-capabilities.json" + +# Colors for output +GREEN='\033[0;32m' +YELLOW='\033[1;33m' +RED='\033[0;31m' +NC='\033[0m' # No Color + +# Parse arguments +PROVIDER_ENV_FILE="" +ARGS=() +while [[ $# -gt 0 ]]; do + case "$1" in + --env) + if [[ -z "${2:-}" || "${2:-}" == --* ]]; then + echo -e "${RED}Error: --env requires a value${NC}" + exit 1 + fi + PROVIDER_ENV_FILE="$2" + shift 2 + ;; + --help) + echo "Usage: $0 [OPTIONS]" + echo "" + echo "Options:" + echo " --env Postman env path or provider name (e.g. openai or provider_config/bifrost-v1-openai.postman_environment.json)" + echo " --verbose Show detailed output" + echo " --html Generate HTML report" + echo " --json Generate JSON report" + echo " --bail Stop on first failure" + echo " --help Show this help message" + echo "" + echo "Environment Variables:" + echo " BIFROST_BASE_URL Override base URL (default: http://localhost:8080)" + echo "" + echo "Examples:" + echo " $0 --env openai # Run with OpenAI provider" + exit 0 + ;; + *) + ARGS+=("$1") + shift + ;; + esac +done +set -- "${ARGS[@]}" + +# Print banner +echo -e "${GREEN}==============================================${NC}" +echo -e "${GREEN}Bifrost V1 Virtual Key Expiry Test Runner${NC}" +echo -e "${GREEN}==============================================${NC}" +echo "" + +# Check if Newman is installed +if ! command -v newman &> /dev/null; then + echo -e "${RED}Error: Newman is not installed${NC}" + echo "Install it with: npm install -g newman" + exit 1 +fi + +# Check if collection exists +if [ ! -f "$COLLECTION" ]; then + echo -e "${RED}Error: Collection file not found: $COLLECTION${NC}" + exit 1 +fi + +# Create report directory +mkdir -p "$REPORT_DIR" + +# Load provider capabilities into globals (for consistency with v1 runner) +GLOBALS_TMP="" +if [ -f "$PROVIDER_CAPABILITIES_JSON" ] && command -v jq &>/dev/null; then + GLOBALS_TMP=$(mktemp) + trap 'rm -f "$GLOBALS_TMP"' EXIT + jq -n --rawfile cap "$PROVIDER_CAPABILITIES_JSON" '{id: "bifrost-provider-capabilities", name: "Provider capabilities", values: [{key: "provider_capabilities", value: $cap, type: "default", enabled: true}]}' > "$GLOBALS_TMP" +fi + +# Parse remaining options +VERBOSE="" +REPORTERS="cli" +BAIL="" +while [[ $# -gt 0 ]]; do + case $1 in + --verbose) + VERBOSE="--verbose" + shift + ;; + --html) + REPORTERS="${REPORTERS},html" + shift + ;; + --json) + REPORTERS="${REPORTERS},json" + shift + ;; + --bail) + BAIL="--bail" + shift + ;; + *) + echo -e "${RED}Unknown option: $1${NC}" + exit 1 + ;; + esac +done + +# Resolve provider env file +SINGLE_JSON_ENV="" +if [ -n "$PROVIDER_ENV_FILE" ]; then + if [ -f "$PROVIDER_ENV_FILE" ]; then + SINGLE_JSON_ENV="$PROVIDER_ENV_FILE" + elif [ -f "$PROVIDER_CONFIG_DIR/$PROVIDER_ENV_FILE" ]; then + SINGLE_JSON_ENV="$PROVIDER_CONFIG_DIR/$PROVIDER_ENV_FILE" + elif [ -f "$PROVIDER_CONFIG_DIR/bifrost-v1-${PROVIDER_ENV_FILE}.postman_environment.json" ]; then + SINGLE_JSON_ENV="$PROVIDER_CONFIG_DIR/bifrost-v1-${PROVIDER_ENV_FILE}.postman_environment.json" + else + echo -e "${RED}Error: Could not find environment file for: $PROVIDER_ENV_FILE${NC}" + echo "Searched:" + echo " - $PROVIDER_ENV_FILE" + echo " - $PROVIDER_CONFIG_DIR/$PROVIDER_ENV_FILE" + echo " - $PROVIDER_CONFIG_DIR/bifrost-v1-${PROVIDER_ENV_FILE}.postman_environment.json" + exit 1 + fi +fi + +# Default to openai if no env specified +if [ -z "$SINGLE_JSON_ENV" ]; then + if [ -f "$PROVIDER_CONFIG_DIR/bifrost-v1-openai.postman_environment.json" ]; then + SINGLE_JSON_ENV="$PROVIDER_CONFIG_DIR/bifrost-v1-openai.postman_environment.json" + echo -e "${YELLOW}No --env specified, using openai${NC}" + fi +fi + +# Build Newman command +cmd=(newman run "$COLLECTION") +[ -n "$GLOBALS_TMP" ] && [ -f "$GLOBALS_TMP" ] && cmd+=(-g "$GLOBALS_TMP") +[ -n "$SINGLE_JSON_ENV" ] && [ -f "$SINGLE_JSON_ENV" ] && cmd+=(-e "$SINGLE_JSON_ENV") + +# Base URL override +base_url="${BIFROST_BASE_URL:-http://localhost:8080}" +cmd+=(--env-var "base_url=$base_url") + +cmd+=(--timeout-script 120000 --timeout 900000) +cmd+=(-r "$REPORTERS") + +if [[ "$REPORTERS" == *"html"* ]]; then + cmd+=(--reporter-html-export "$REPORT_DIR/report.html") +fi +if [[ "$REPORTERS" == *"json"* ]]; then + cmd+=(--reporter-json-export "$REPORT_DIR/report.json") +fi +[ -n "$VERBOSE" ] && cmd+=("$VERBOSE") +[ -n "$BAIL" ] && cmd+=("$BAIL") + +echo -e "Configuration:" +echo -e " Collection: ${YELLOW}$COLLECTION${NC}" +echo -e " Base URL: ${YELLOW}$base_url${NC}" +if [ -n "$SINGLE_JSON_ENV" ]; then + echo -e " Env: ${YELLOW}$SINGLE_JSON_ENV${NC}" +fi +echo -e " Reports: ${YELLOW}$REPORT_DIR${NC}" +echo "" +echo -e "${GREEN}Running tests...${NC}" +echo "" + +set +e +"${cmd[@]}" +EXIT_CODE=$? +set -e + +echo "" +if [ $EXIT_CODE -eq 0 ]; then + echo -e "${GREEN}✓ All VK expiry tests passed!${NC}" +else + echo -e "${RED}✗ Some tests failed${NC}" +fi +if [[ "$REPORTERS" == *"html"* ]] || [[ "$REPORTERS" == *"json"* ]]; then + echo "" + echo -e "Reports saved to: ${YELLOW}$REPORT_DIR${NC}" + ls -lh "$REPORT_DIR" 2>/dev/null | tail -n +2 +fi + +exit $EXIT_CODE diff --git a/tests/e2e/clis/.gitignore b/tests/e2e/clis/.gitignore deleted file mode 100644 index 573b3a8b4ab..00000000000 --- a/tests/e2e/clis/.gitignore +++ /dev/null @@ -1,2 +0,0 @@ -reports/* -!reports/.keep diff --git a/tests/e2e/clis/reports/.keep b/tests/e2e/clis/reports/.keep deleted file mode 100644 index e69de29bb2d..00000000000 diff --git a/tests/e2e/features/virtual-keys/pages/virtual-keys.page.ts b/tests/e2e/features/virtual-keys/pages/virtual-keys.page.ts index 593c9151543..72ba83382fe 100644 --- a/tests/e2e/features/virtual-keys/pages/virtual-keys.page.ts +++ b/tests/e2e/features/virtual-keys/pages/virtual-keys.page.ts @@ -20,6 +20,7 @@ const PROVIDER_DISPLAY_NAMES: Record = { openrouter: "OpenRouter", huggingface: "HuggingFace", cerebras: "Cerebras", + deepseek: "DeepSeek", perplexity: "Perplexity", elevenlabs: "Elevenlabs", parasail: "Parasail", @@ -874,4 +875,4 @@ export class VirtualKeysPage extends BasePage { } } } -} \ No newline at end of file +} diff --git a/tests/semanticcache/.gitignore b/tests/semanticcache/.gitignore deleted file mode 100644 index 6c7f6431d4b..00000000000 --- a/tests/semanticcache/.gitignore +++ /dev/null @@ -1,2 +0,0 @@ -reports/ -*.log diff --git a/transports/bifrost-http/handlers/config.go b/transports/bifrost-http/handlers/config.go index 4a3497502c7..e934f3c9026 100644 --- a/transports/bifrost-http/handlers/config.go +++ b/transports/bifrost-http/handlers/config.go @@ -253,7 +253,7 @@ func (h *ConfigHandler) updateMetadata(ctx *fasthttp.RequestCtx) { } var patch map[string]any if err := json.Unmarshal(ctx.PostBody(), &patch); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("invalid request format: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } if len(patch) == 0 { @@ -287,7 +287,7 @@ func (h *ConfigHandler) updateConfig(ctx *fasthttp.RequestCtx) { }{} if err := json.Unmarshal(ctx.PostBody(), &payload); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid request format: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } @@ -342,6 +342,58 @@ func (h *ConfigHandler) updateConfig(ctx *fasthttp.RequestCtx) { currentConfig := h.store.ClientConfig updatedConfig := currentConfig + // Validate MCP auth-mode / OAuth2 server settings before any live mutation + // below (drop-excess flag, MCP tool-manager reload, compat plugin reload, + // in-memory MCP config). A late rejection would return 400 while runtime + // state had already changed but DB persistence was skipped, diverging + // in-memory, core, and DB state. + + // Validate the inbound MCP auth mode against the allowed enum + // (config.schema.json is the source of truth: headers | both | oauth). + switch payload.ClientConfig.MCPServerAuthMode { + case "", configstoreTables.MCPServerAuthModeHeaders, configstoreTables.MCPServerAuthModeBoth, configstoreTables.MCPServerAuthModeOAuth: + // valid; empty means the field was omitted from a partial update + default: + SendError(ctx, fasthttp.StatusBadRequest, "mcp_server_auth_mode must be one of: headers, both, oauth") + return + } + + // oauth2_server_config only applies when discovery is enabled (both | oauth). + // Evaluate against the effective mode so a partial update that supplies only + // the config cannot smuggle it in while the stored mode is headers. + effectiveAuthMode := payload.ClientConfig.MCPServerAuthMode + if effectiveAuthMode == "" { + effectiveAuthMode = currentConfig.MCPServerAuthMode + } + effectiveOAuth2Config := currentConfig.OAuth2ServerConfig + if payload.ClientConfig.OAuth2ServerConfig != nil { + effectiveOAuth2Config = payload.ClientConfig.OAuth2ServerConfig + } + + // disable_vk_identity only makes sense in oauth mode: in both mode virtual + // keys can still authenticate via headers, so suppressing them in the consent + // flow alone would be misleading. Evaluate the merged config so a partial + // update that switches the mode away from oauth (without resending the config) + // cannot leave a previously stored disable_vk_identity active. + if effectiveOAuth2Config != nil && + effectiveOAuth2Config.DisableVKIdentity && + effectiveAuthMode != configstoreTables.MCPServerAuthModeOAuth { + SendError(ctx, fasthttp.StatusBadRequest, "disable_vk_identity is only valid when mcp_server_auth_mode is oauth") + return + } + + // Cap auth_code_ttl so a leaked one-time code can't stay valid for long. + // This is an unconditional invariant on the stored value — enforced in every + // mode (not just both | oauth), mirroring the load-time validateClientConfig + // check — so a save can never persist a value that would then fail boot on the + // next restart. A zero/omitted value falls back to the default at issuance and + // is left alone here. + if effectiveOAuth2Config != nil && + effectiveOAuth2Config.AuthCodeTTL > configstoreTables.MaxAuthCodeTTL { + SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("auth_code_ttl must not exceed %d seconds (15 minutes)", configstoreTables.MaxAuthCodeTTL)) + return + } + var restartReasons []string if payload.ClientConfig.DropExcessRequests != currentConfig.DropExcessRequests { @@ -534,10 +586,22 @@ func (h *ConfigHandler) updateConfig(ctx *fasthttp.RequestCtx) { updatedConfig.RoutingChainMaxDepth = payload.ClientConfig.RoutingChainMaxDepth } - // Update external base URLs for OAuth server metadata and client redirect_uri (nil clears each override). + // Update external base URL for OAuth client redirect_uri (nil clears the override). // Validation is performed up front in this handler so a failure here cannot leave the process in a partial state. updatedConfig.MCPExternalClientURL = payload.ClientConfig.MCPExternalClientURL + // Only update each field when explicitly provided so partial /api/config + // payloads do not clear stored values (matches the MCP field handling above). + // The enum, disable_vk_identity, and auth_code_ttl validations for these + // fields run up front (before any live mutation) so a rejection can't leave + // runtime and DB state diverged. + if payload.ClientConfig.MCPServerAuthMode != "" { + updatedConfig.MCPServerAuthMode = payload.ClientConfig.MCPServerAuthMode + } + if payload.ClientConfig.OAuth2ServerConfig != nil { + updatedConfig.OAuth2ServerConfig = payload.ClientConfig.OAuth2ServerConfig + } + // Handle HeaderFilterConfig changes if !headerFilterConfigEqual(payload.ClientConfig.HeaderFilterConfig, currentConfig.HeaderFilterConfig) { // Validate that no security headers are in the allowlist or denylist @@ -885,7 +949,7 @@ func (h *ConfigHandler) updateProxyConfig(ctx *fasthttp.RequestCtx) { var payload configstoreTables.GlobalProxyConfig if err := json.Unmarshal(ctx.PostBody(), &payload); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("invalid request format: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } diff --git a/transports/bifrost-http/handlers/featureflags.go b/transports/bifrost-http/handlers/featureflags.go index 27d9556d242..8a350409043 100644 --- a/transports/bifrost-http/handlers/featureflags.go +++ b/transports/bifrost-http/handlers/featureflags.go @@ -65,7 +65,7 @@ func (h *FeatureFlagsHandler) updateFlag(ctx *fasthttp.RequestCtx) { var req updateFlagRequest if err := sonic.Unmarshal(ctx.PostBody(), &req); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, "Invalid request body: "+err.Error()) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } if req.Enabled == nil { diff --git a/transports/bifrost-http/handlers/governance.go b/transports/bifrost-http/handlers/governance.go index 64e1d2dde35..cf3b028ebb6 100644 --- a/transports/bifrost-http/handlers/governance.go +++ b/transports/bifrost-http/handlers/governance.go @@ -10,6 +10,7 @@ import ( "fmt" "io" "math" + "net/url" "sort" "strconv" "strings" @@ -105,6 +106,20 @@ func lookupScopeNameResolver(scope string) (ScopeNameResolver, bool) { return fn, ok } +// ExternalQuotaBudgetResolver returns budgets that govern a VK but whose usage is +// tracked OUTSIDE the VK's own budget rows +type ExternalQuotaBudgetResolver func(ctx context.Context, vk *configstoreTables.TableVirtualKey) (*ExternalQuotaBudgetResult, error) + +// ExternalQuotaBudgetResult is what an external resolver returns for a VK whose +// authoritative usage lives outside its own budget rows. +type ExternalQuotaBudgetResult struct { + // Budgets replaces the VK's own budget rows in the quota response. + Budgets []configstoreTables.TableBudget + // UsageUserID, when non-empty, scopes the per_model_usage query to this user id + // instead of the VK id. + UsageUserID string +} + type GovernanceHandler struct { configStore configstore.ConfigStore governanceManager GovernanceManager @@ -112,15 +127,22 @@ type GovernanceHandler struct { // endpoint's model_usage breakdown. Optional: nil when the logging plugin is // not enabled, in which case the breakdown is simply omitted. logManager logging.LogManager + // externalQuotaBudgetResolver, when non-nil, supplies budgets that govern a VK + // but whose usage lives outside the VK's own budget rows (enterprise + // access-profile-managed VKs). Injected at construction; nil on OSS builds. + externalQuotaBudgetResolver ExternalQuotaBudgetResolver } // NewGovernanceHandler creates a new governance handler instance. // logManager is optional (may be nil); when supplied it powers the quota // endpoint's per-budget actual per-model usage breakdown. +// externalQuotaBudgetResolver is optional (may be nil); when supplied the quota +// endpoint uses it to resolve budgets/usage for VKs whose authoritative usage +// is tracked outside their own budget rows. // Side effect: ensures the default virtual_key scope-name resolver is // registered against the supplied configStore, so resolveModelConfigScopeName // can render VK names for OSS-only builds without further wiring. -func NewGovernanceHandler(manager GovernanceManager, configStore configstore.ConfigStore, logManager logging.LogManager) (*GovernanceHandler, error) { +func NewGovernanceHandler(manager GovernanceManager, configStore configstore.ConfigStore, logManager logging.LogManager, externalQuotaBudgetResolver ExternalQuotaBudgetResolver) (*GovernanceHandler, error) { if manager == nil { return nil, fmt.Errorf("governance manager is required") } @@ -135,9 +157,10 @@ func NewGovernanceHandler(manager GovernanceManager, configStore configstore.Con return vk.Name, true }) return &GovernanceHandler{ - governanceManager: manager, - configStore: configStore, - logManager: logManager, + governanceManager: manager, + configStore: configStore, + logManager: logManager, + externalQuotaBudgetResolver: externalQuotaBudgetResolver, }, nil } @@ -164,6 +187,7 @@ type CreateVirtualKeyRequest struct { RateLimit *CreateRateLimitRequest `json:"rate_limit,omitempty"` IsActive *bool `json:"is_active,omitempty"` CalendarAligned bool `json:"calendar_aligned,omitempty"` // When true, all budgets reset at clean calendar boundaries + ExpiresAt *time.Time `json:"expires_at,omitempty"` // Optional expiry; nil means never expires } // UpdateVirtualKeyRequest represents the request body for updating a virtual key @@ -192,6 +216,7 @@ type UpdateVirtualKeyRequest struct { IsActive *bool `json:"is_active,omitempty"` CalendarAligned *bool `json:"calendar_aligned,omitempty"` // When true, all budgets reset at clean calendar boundaries ResetBudgetUsage *bool `json:"reset_budget_usage,omitempty"` + ExpiresAt *string `json:"expires_at,omitempty"` // RFC3339 timestamp sets a new expiry, "" clears it, omitted leaves it unchanged } var errVirtualKeyDualAssociation = errors.New("VirtualKey cannot be attached to both Team and Customer") @@ -1051,7 +1076,7 @@ func (h *GovernanceHandler) updateComplexityAnalyzerConfig(ctx *fasthttp.Request decoder := json.NewDecoder(bytes.NewReader(ctx.PostBody())) decoder.DisallowUnknownFields() if err := decoder.Decode(&payload); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("invalid request format: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } if err := decoder.Decode(&struct{}{}); err != io.EOF { @@ -1270,6 +1295,14 @@ func (h *GovernanceHandler) createVirtualKey(ctx *fasthttp.RequestCtx) { seenDurations[b.ResetDuration] = true } } + // Validate expires_at: must be in the future if provided + if req.ExpiresAt != nil { + now := time.Now().UTC() + if !req.ExpiresAt.After(now) { + SendError(ctx, 400, "expires_at must be a future timestamp") + return + } + } // Set defaults: nil means "use DB default (true)" isActive := req.IsActive if isActive == nil { @@ -1290,12 +1323,13 @@ func (h *GovernanceHandler) createVirtualKey(ctx *fasthttp.RequestCtx) { vk = configstoreTables.TableVirtualKey{ ID: uuid.NewString(), Name: req.Name, - Value: governance.GenerateVirtualKey(), + Value: *schemas.NewSecretVar(governance.GenerateVirtualKey()), Description: req.Description, TeamID: req.TeamID, CustomerID: req.CustomerID, IsActive: isActive, CalendarAligned: req.CalendarAligned, + ExpiresAt: req.ExpiresAt, } if err := h.configStore.CreateVirtualKey(ctx, &vk, tx); err != nil { return err @@ -1494,6 +1528,20 @@ func (h *GovernanceHandler) updateVirtualKey(ctx *fasthttp.RequestCtx) { SendError(ctx, 400, "VirtualKey cannot be attached to both Team and Customer") return } + // Parse expires_at when provided: a timestamp must be in the future, "" clears the expiry. + var newExpiresAt *time.Time + if req.ExpiresAt != nil && *req.ExpiresAt != "" { + parsed, err := time.Parse(time.RFC3339, *req.ExpiresAt) + if err != nil { + SendError(ctx, 400, "expires_at must be an RFC3339 timestamp") + return + } + if !parsed.After(time.Now().UTC()) { + SendError(ctx, 400, "expires_at must be a future timestamp") + return + } + newExpiresAt = &parsed + } vk, err := h.configStore.GetVirtualKey(ctx, vkID) if err != nil { if errors.Is(err, configstore.ErrNotFound) { @@ -1550,6 +1598,9 @@ func (h *GovernanceHandler) updateVirtualKey(ctx *fasthttp.RequestCtx) { if req.IsActive != nil { vk.IsActive = req.IsActive } + if req.ExpiresAt != nil { + vk.ExpiresAt = newExpiresAt + } if req.CalendarAligned != nil { vk.CalendarAligned = *req.CalendarAligned } @@ -1903,9 +1954,9 @@ func (h *GovernanceHandler) rotateVirtualKeyByID(ctx context.Context, vkID strin if err != nil { return nil, err } - oldValue := vk.Value - vk.Value = governance.GenerateVirtualKey() - if vk.Value == oldValue { + oldValue := vk.Value.GetValue() + vk.Value = *schemas.NewSecretVar(governance.GenerateVirtualKey()) + if vk.Value.GetValue() == oldValue { return nil, fmt.Errorf("generated virtual key matched existing value") } if err := h.configStore.UpdateVirtualKey(ctx, vk); err != nil { @@ -2193,7 +2244,13 @@ func (h *GovernanceHandler) createTeam(ctx *fasthttp.RequestCtx) { // getTeam handles GET /api/governance/teams/{team_id} - Get a specific team func (h *GovernanceHandler) getTeam(ctx *fasthttp.RequestCtx) { - teamID := ctx.UserValue("team_id").(string) + // The router matches on the raw (percent-encoded) path, so SCIM/IdP-synced team + // IDs containing spaces or other URL-sensitive characters arrive still encoded. + teamID, err := url.PathUnescape(ctx.UserValue("team_id").(string)) + if err != nil { + SendError(ctx, 400, "Invalid team ID encoding") + return + } team, err := h.configStore.GetTeam(ctx, teamID) if err != nil { if errors.Is(err, configstore.ErrNotFound) { @@ -2210,7 +2267,13 @@ func (h *GovernanceHandler) getTeam(ctx *fasthttp.RequestCtx) { // updateTeam handles PUT /api/governance/teams/{team_id} - Update a team func (h *GovernanceHandler) updateTeam(ctx *fasthttp.RequestCtx) { - teamID := ctx.UserValue("team_id").(string) + // The router matches on the raw (percent-encoded) path, so SCIM/IdP-synced team + // IDs containing spaces or other URL-sensitive characters arrive still encoded. + teamID, err := url.PathUnescape(ctx.UserValue("team_id").(string)) + if err != nil { + SendError(ctx, 400, "Invalid team ID encoding") + return + } var req UpdateTeamRequest if err := json.Unmarshal(ctx.PostBody(), &req); err != nil { @@ -2448,7 +2511,13 @@ func (h *GovernanceHandler) updateTeam(ctx *fasthttp.RequestCtx) { // deleteTeam handles DELETE /api/governance/teams/{team_id} - Delete a team func (h *GovernanceHandler) deleteTeam(ctx *fasthttp.RequestCtx) { - teamID := ctx.UserValue("team_id").(string) + // The router matches on the raw (percent-encoded) path, so SCIM/IdP-synced team + // IDs containing spaces or other URL-sensitive characters arrive still encoded. + teamID, err := url.PathUnescape(ctx.UserValue("team_id").(string)) + if err != nil { + SendError(ctx, 400, "Invalid team ID encoding") + return + } team, err := h.configStore.GetTeam(ctx, teamID) if err != nil { if errors.Is(err, configstore.ErrNotFound) { @@ -3514,7 +3583,11 @@ func (h *GovernanceHandler) getProviderGovernance(ctx *fasthttp.RequestCtx) { // updateProviderGovernance handles PUT /api/governance/providers/{provider_name} - Update provider governance func (h *GovernanceHandler) updateProviderGovernance(ctx *fasthttp.RequestCtx) { - providerName := ctx.UserValue("provider_name").(string) + providerName, err := url.PathUnescape(ctx.UserValue("provider_name").(string)) + if err != nil { + SendError(ctx, 400, "Invalid provider name encoding") + return + } var req UpdateProviderGovernanceRequest if err := json.Unmarshal(ctx.PostBody(), &req); err != nil { SendError(ctx, 400, "Invalid JSON") @@ -3756,7 +3829,11 @@ func (h *GovernanceHandler) updateProviderGovernance(ctx *fasthttp.RequestCtx) { // deleteProviderGovernance handles DELETE /api/governance/providers/{provider_name} - removes // provider-level governance by deleting the all-models model config for that provider. func (h *GovernanceHandler) deleteProviderGovernance(ctx *fasthttp.RequestCtx) { - providerName := ctx.UserValue("provider_name").(string) + providerName, err := url.PathUnescape(ctx.UserValue("provider_name").(string)) + if err != nil { + SendError(ctx, 400, "Invalid provider name encoding") + return + } mc, err := h.configStore.GetModelConfig(ctx, configstoreTables.ModelConfigScopeGlobal, nil, configstoreTables.ModelConfigAllModels, &providerName) if err != nil { if err == configstore.ErrNotFound { @@ -4648,29 +4725,28 @@ type quotaBudget struct { Models []quotaModelSpend `json:"per_model_usage"` } -// buildVKBudgetsWithUsage wraps each hydrated VK budget with its per-model actual usage, -// queried from request logs over that budget's current cycle [last_reset, now]. Per-budget -// because a VK's budgets can have independent reset cycles (e.g. daily + monthly). When -// logging is disabled (logManager == nil) the budgets are returned with an empty models list -// — that is the only case where per_model_usage is empty. A log-store query failure instead -// returns an error so the endpoint fails closed (500) rather than reporting empty usage that -// callers cannot distinguish from "logging disabled". Callers must hydrate vk.Budgets (via -// collectVKModelUsage) before calling this. -func (h *GovernanceHandler) buildVKBudgetsWithUsage(ctx context.Context, vk *configstoreTables.TableVirtualKey, now time.Time) ([]quotaBudget, error) { - out := make([]quotaBudget, 0, len(vk.Budgets)) - for i := range vk.Budgets { - b := &vk.Budgets[i] +// buildBudgetsWithUsage wraps each budget with its per-model actual usage, queried from +// request logs over that budget's current cycle +func (h *GovernanceHandler) buildBudgetsWithUsage(ctx context.Context, vkID, usageUserID string, budgets []configstoreTables.TableBudget, now time.Time) ([]quotaBudget, error) { + out := make([]quotaBudget, 0, len(budgets)) + for i := range budgets { + b := &budgets[i] entry := quotaBudget{TableBudget: *b, Models: []quotaModelSpend{}} if h.logManager != nil { start := b.LastReset if b.CreatedAt.After(start) { start = b.CreatedAt } - ranking, err := h.logManager.GetModelRankings(ctx, &logstore.SearchFilters{ - VirtualKeyIDs: []string{vk.ID}, - StartTime: &start, - EndTime: &now, - }) + filters := &logstore.SearchFilters{ + StartTime: &start, + EndTime: &now, + } + if usageUserID != "" { + filters.UserIDs = []string{usageUserID} + } else { + filters.VirtualKeyIDs = []string{vkID} + } + ranking, err := h.logManager.GetModelRankings(ctx, filters) if err != nil { logger.Error("failed to load per-model usage for VK quota (budget %s): %v", b.ID, err) return nil, err @@ -4728,9 +4804,23 @@ func (h *GovernanceHandler) getVirtualKeyQuota(ctx *fasthttp.RequestCtx) { return } + budgetRows := vk.Budgets + usageUserID := "" + if resolve := h.externalQuotaBudgetResolver; resolve != nil { + ext, err := resolve(ctx, vk) + if err != nil { + SendError(ctx, 500, "Failed to load access-profile usage") + return + } + if ext != nil { + budgetRows = ext.Budgets + usageUserID = ext.UsageUserID + } + } + // Each budget carries its actual per-model spend (from request logs) for the current // cycle. Must run after collectVKModelUsage, which hydrates vk.Budgets. - budgets, err := h.buildVKBudgetsWithUsage(ctx, vk, time.Now()) + budgets, err := h.buildBudgetsWithUsage(ctx, vk.ID, usageUserID, budgetRows, time.Now()) if err != nil { SendError(ctx, 500, "Failed to load per-model usage") return diff --git a/transports/bifrost-http/handlers/governance_test.go b/transports/bifrost-http/handlers/governance_test.go index 69a1b666853..08f6b1ed6aa 100644 --- a/transports/bifrost-http/handlers/governance_test.go +++ b/transports/bifrost-http/handlers/governance_test.go @@ -27,11 +27,12 @@ import ( type mockGovernanceManagerForVK struct { GovernanceManager getGovernanceDataCalls int + data *governance.GovernanceData } func (m *mockGovernanceManagerForVK) GetGovernanceData(ctx context.Context) *governance.GovernanceData { m.getGovernanceDataCalls++ - return nil + return m.data } // mockConfigStoreForVK embeds the interface so unimplemented methods panic. @@ -232,7 +233,7 @@ func TestComplexityAnalyzerConfigPutRejectsInvalidPayloads(t *testing.T) { body string want string }{ - {name: "unknown field", body: strings.TrimSuffix(validBody, "}") + `,"extra":true}`, want: "unknown field"}, + {name: "unknown field", body: strings.TrimSuffix(validBody, "}") + `,"extra":true}`, want: "Invalid request payload"}, {name: "multiple json values", body: validBody + `{}`, want: "multiple JSON values"}, {name: "invalid boundaries", body: testComplexityAnalyzerPayload(t, invalidBoundaries), want: "tier boundaries"}, {name: "empty keywords", body: testComplexityAnalyzerPayload(t, emptyKeywords), want: "keyword lists must be non-empty"}, @@ -1040,7 +1041,7 @@ func TestRotateVirtualKey_OnlyChangesValueAndReloads(t *testing.T) { "vk-1": { ID: "vk-1", Name: "Production", - Value: "sk-bf-old", + Value: *schemas.NewSecretVar("sk-bf-old"), Description: "existing description", TeamID: &teamID, RateLimitID: &rateLimitID, @@ -1076,11 +1077,11 @@ func TestRotateVirtualKey_OnlyChangesValueAndReloads(t *testing.T) { } updated := store.virtualKeys["vk-1"] - if updated.Value == "sk-bf-old" { + if updated.Value.GetValue() == "sk-bf-old" { t.Fatal("expected virtual key value to rotate") } - if !strings.HasPrefix(updated.Value, governance.VirtualKeyPrefix) { - t.Fatalf("expected rotated value to use %q prefix, got %q", governance.VirtualKeyPrefix, updated.Value) + if !strings.HasPrefix(updated.Value.GetValue(), governance.VirtualKeyPrefix) { + t.Fatalf("expected rotated value to use %q prefix, got %q", governance.VirtualKeyPrefix, updated.Value.GetValue()) } if updated.ID != "vk-1" || updated.Name != "Production" || updated.Description != "existing description" { t.Fatalf("rotation changed non-value fields: %#v", updated) @@ -1105,8 +1106,8 @@ func TestRotateVirtualKey_OnlyChangesValueAndReloads(t *testing.T) { if err := json.Unmarshal(ctx.Response.Body(), &resp); err != nil { t.Fatalf("failed to parse response: %v", err) } - if resp.VirtualKey.Value != updated.Value { - t.Fatalf("response value = %q, want %q", resp.VirtualKey.Value, updated.Value) + if resp.VirtualKey.Value.GetValue() != updated.Value.GetValue() { + t.Fatalf("response value = %q, want %q", resp.VirtualKey.Value.GetValue(), updated.Value.GetValue()) } } @@ -1138,7 +1139,7 @@ func TestRotateVirtualKey_UpdateFailureDoesNotReload(t *testing.T) { store := &mockRotateConfigStore{ virtualKeys: map[string]*configstoreTables.TableVirtualKey{ - "vk-1": {ID: "vk-1", Name: "One", Value: "sk-bf-old"}, + "vk-1": {ID: "vk-1", Name: "One", Value: *schemas.NewSecretVar("sk-bf-old")}, }, updateErr: errors.New("database unavailable"), } @@ -1153,8 +1154,8 @@ func TestRotateVirtualKey_UpdateFailureDoesNotReload(t *testing.T) { if ctx.Response.StatusCode() != 500 { t.Fatalf("expected status 500, got %d: %s", ctx.Response.StatusCode(), string(ctx.Response.Body())) } - if store.virtualKeys["vk-1"].Value != "sk-bf-old" { - t.Fatalf("expected value to remain unchanged, got %q", store.virtualKeys["vk-1"].Value) + if store.virtualKeys["vk-1"].Value.GetValue() != "sk-bf-old" { + t.Fatalf("expected value to remain unchanged, got %q", store.virtualKeys["vk-1"].Value.GetValue()) } if len(manager.reloadIDs) != 0 { t.Fatalf("expected no reloads, got %#v", manager.reloadIDs) @@ -1166,7 +1167,7 @@ func TestRotateVirtualKey_ReloadFailureReturnsErrorAfterUpdate(t *testing.T) { store := &mockRotateConfigStore{ virtualKeys: map[string]*configstoreTables.TableVirtualKey{ - "vk-1": {ID: "vk-1", Name: "One", Value: "sk-bf-old"}, + "vk-1": {ID: "vk-1", Name: "One", Value: *schemas.NewSecretVar("sk-bf-old")}, }, } manager := &mockRotateGovernanceManager{store: store, reloadErr: errors.New("reload failed")} @@ -1183,7 +1184,7 @@ func TestRotateVirtualKey_ReloadFailureReturnsErrorAfterUpdate(t *testing.T) { if store.updates != 1 { t.Fatalf("expected one update, got %d", store.updates) } - if store.virtualKeys["vk-1"].Value == "sk-bf-old" { + if store.virtualKeys["vk-1"].Value.GetValue() == "sk-bf-old" { t.Fatal("expected value to rotate before reload failure") } if len(manager.reloadIDs) != 1 || manager.reloadIDs[0] != "vk-1" { @@ -1199,8 +1200,8 @@ func TestRotateVirtualKeys_PartialSuccess(t *testing.T) { store := &mockRotateConfigStore{ virtualKeys: map[string]*configstoreTables.TableVirtualKey{ - "vk-1": {ID: "vk-1", Name: "One", Value: "sk-bf-old-1"}, - "vk-2": {ID: "vk-2", Name: "Two", Value: "sk-bf-old-2"}, + "vk-1": {ID: "vk-1", Name: "One", Value: *schemas.NewSecretVar("sk-bf-old-1")}, + "vk-2": {ID: "vk-2", Name: "Two", Value: *schemas.NewSecretVar("sk-bf-old-2")}, }, } manager := &mockRotateGovernanceManager{store: store} @@ -1220,7 +1221,7 @@ func TestRotateVirtualKeys_PartialSuccess(t *testing.T) { if len(manager.reloadIDs) != 2 || manager.reloadIDs[0] != "vk-1" || manager.reloadIDs[1] != "vk-2" { t.Fatalf("expected reloads for vk-1 and vk-2, got %#v", manager.reloadIDs) } - if store.virtualKeys["vk-1"].Value == "sk-bf-old-1" || store.virtualKeys["vk-2"].Value == "sk-bf-old-2" { + if store.virtualKeys["vk-1"].Value.GetValue() == "sk-bf-old-1" || store.virtualKeys["vk-2"].Value.GetValue() == "sk-bf-old-2" { t.Fatalf("expected successful IDs to rotate: %#v", store.virtualKeys) } @@ -1256,7 +1257,7 @@ func TestRotateVirtualKeys_RejectsInvalidRequests(t *testing.T) { t.Run(tt.name, func(t *testing.T) { store := &mockRotateConfigStore{ virtualKeys: map[string]*configstoreTables.TableVirtualKey{ - "vk-1": {ID: "vk-1", Name: "One", Value: "sk-bf-old-1"}, + "vk-1": {ID: "vk-1", Name: "One", Value: *schemas.NewSecretVar("sk-bf-old-1")}, }, } manager := &mockRotateGovernanceManager{store: store} @@ -1288,8 +1289,8 @@ func TestRotateVirtualKeys_TrimsAndDeduplicatesIDs(t *testing.T) { store := &mockRotateConfigStore{ virtualKeys: map[string]*configstoreTables.TableVirtualKey{ - "vk-1": {ID: "vk-1", Name: "One", Value: "sk-bf-old-1"}, - "vk-2": {ID: "vk-2", Name: "Two", Value: "sk-bf-old-2"}, + "vk-1": {ID: "vk-1", Name: "One", Value: *schemas.NewSecretVar("sk-bf-old-1")}, + "vk-2": {ID: "vk-2", Name: "Two", Value: *schemas.NewSecretVar("sk-bf-old-2")}, }, } manager := &mockRotateGovernanceManager{store: store} @@ -1571,6 +1572,122 @@ func TestGetVirtualKeyQuota_HydratesBudgetsFromModelConfigs(t *testing.T) { } } +// TestGetVirtualKeyQuota_ExternalResolverReplacesWithAccessProfileBudgets verifies the +// AP-managed-VK path: when a registered ExternalQuotaBudgetResolver returns budgets +// (enterprise access-profile budgets, which carry the real usage), they REPLACE the VK's +// own budget rows in the quota response. Those rows are reset to current_usage=0 at +// adoption and never charged again, so reporting them would be a misleading $0 row. +func TestGetVirtualKeyQuota_ExternalResolverReplacesWithAccessProfileBudgets(t *testing.T) { + SetLogger(&mockLogger{}) + + active := true + cycleStart := time.Date(2026, time.January, 2, 15, 4, 5, 0, time.UTC) + store := &mockQuotaConfigStore{ + vk: &configstoreTables.TableVirtualKey{ + ID: "vk-1", + Name: "AP Key", + IsActive: &active, + }, + modelConfigs: []configstoreTables.TableModelConfig{ + { + ID: "mc-vk", + Scope: configstoreTables.ModelConfigScopeVirtualKey, + ScopeID: schemas.Ptr("vk-1"), + ModelName: configstoreTables.ModelConfigAllModels, + // VK mirror row: zero usage, reset at adoption — must NOT appear in the response. + Budgets: []configstoreTables.TableBudget{ + {ID: "b-vk", MaxLimit: 100, CurrentUsage: 0, ResetDuration: "1d", LastReset: cycleStart}, + }, + }, + }, + } + // The AP user's inference is logged under user-1 (virtual_key_id is empty on SSO/AP + // log rows), so the per-model usage query must be scoped to the user, not the VK. + logMgr := &mockQuotaLogManager{ + rankings: &logstore.ModelRankingResult{ + Rankings: []logstore.ModelRankingWithTrend{ + {ModelRankingEntry: logstore.ModelRankingEntry{Model: "claude-opus-4-7", Provider: "anthropic", TotalRequests: 2, TotalTokens: 900, TotalCost: 42}}, + }, + }, + } + h := &GovernanceHandler{ + configStore: store, + logManager: logMgr, + externalQuotaBudgetResolver: func(_ context.Context, vk *configstoreTables.TableVirtualKey) (*ExternalQuotaBudgetResult, error) { + if vk.ID != "vk-1" { + return nil, nil + } + return &ExternalQuotaBudgetResult{ + // The access-profile budget that holds the real ongoing usage. + Budgets: []configstoreTables.TableBudget{ + {ID: "b-ap", MaxLimit: 500, CurrentUsage: 42, ResetDuration: "1d", LastReset: cycleStart}, + }, + UsageUserID: "user-1", + }, nil + }, + } + + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("x-bf-vk", "sk-bf-secret") + + h.getVirtualKeyQuota(ctx) + + if ctx.Response.StatusCode() != 200 { + t.Fatalf("expected status 200, got %d: %s", ctx.Response.StatusCode(), string(ctx.Response.Body())) + } + var resp quotaResponse + if err := json.Unmarshal(ctx.Response.Body(), &resp); err != nil { + t.Fatalf("failed to parse response: %v", err) + } + // Only the access-profile budget — the VK's own b-vk row is replaced, not appended. + if len(resp.Budgets) != 1 || resp.Budgets[0].ID != "b-ap" || resp.Budgets[0].CurrentUsage != 42 { + t.Fatalf("expected only the access-profile budget b-ap (usage 42), got %#v", resp.Budgets) + } + // Per-model spend on that budget comes from the user-scoped log query and reconciles + // with current_usage (both 42). + if len(resp.Budgets[0].Models) != 1 || resp.Budgets[0].Models[0].Model != "claude-opus-4-7" || resp.Budgets[0].Models[0].TotalCost != 42 { + t.Fatalf("expected per-model spend from user-scoped logs, got %#v", resp.Budgets[0].Models) + } + // The usage query must be scoped to the AP user, NOT the VK (whose logs are empty). + if len(logMgr.calls) != 1 { + t.Fatalf("expected GetModelRankings called once, got %d", len(logMgr.calls)) + } + call := logMgr.calls[0] + if len(call.UserIDs) != 1 || call.UserIDs[0] != "user-1" { + t.Fatalf("expected usage query scoped to user-1, got UserIDs=%#v", call.UserIDs) + } + if len(call.VirtualKeyIDs) != 0 { + t.Fatalf("expected no VK scoping on the AP usage query, got %#v", call.VirtualKeyIDs) + } +} + +// TestGetVirtualKeyQuota_ExternalResolverErrorFailsClosed verifies the endpoint returns +// 500 (not a partial response) when the registered resolver errors — usage must not be +// silently under-reported. +func TestGetVirtualKeyQuota_ExternalResolverErrorFailsClosed(t *testing.T) { + SetLogger(&mockLogger{}) + + active := true + store := &mockQuotaConfigStore{ + vk: &configstoreTables.TableVirtualKey{ID: "vk-1", Name: "AP Key", IsActive: &active}, + } + h := &GovernanceHandler{ + configStore: store, + externalQuotaBudgetResolver: func(_ context.Context, _ *configstoreTables.TableVirtualKey) (*ExternalQuotaBudgetResult, error) { + return nil, errors.New("boom") + }, + } + + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("x-bf-vk", "sk-bf-secret") + + h.getVirtualKeyQuota(ctx) + + if ctx.Response.StatusCode() != 500 { + t.Fatalf("expected status 500 on resolver error, got %d: %s", ctx.Response.StatusCode(), string(ctx.Response.Body())) + } +} + // TestGetVirtualKeyQuota_NoGovernanceReturnsEmpty verifies that a VK without any // VK-scoped model configs reports empty governance (not a stale direct-relationship // read) and still returns 200 with identity fields. @@ -1742,7 +1859,7 @@ func TestGetVirtualKeyQuota_EndToEndWithRealStore(t *testing.T) { vk := &configstoreTables.TableVirtualKey{ ID: vkID, Name: "Prod", - Value: "sk-bf-e2e-secret", + Value: *schemas.NewSecretVar("sk-bf-e2e-secret"), IsActive: &active, ProviderConfigs: []configstoreTables.TableVirtualKeyProviderConfig{ {VirtualKeyID: vkID, Provider: "openai", AllowAllKeys: true, AllowedModels: schemas.WhiteList{"*"}}, @@ -1908,7 +2025,7 @@ func TestGetVirtualKeyQuota_WindowClampedToBudgetCreation(t *testing.T) { vk := &configstoreTables.TableVirtualKey{ ID: vkID, Name: "Clamp", - Value: "sk-bf-clamp-secret", + Value: *schemas.NewSecretVar("sk-bf-clamp-secret"), IsActive: &active, } if err := store.CreateVirtualKey(ctx, vk); err != nil { @@ -2080,13 +2197,18 @@ func TestGetVirtualKeys_PaginatedEndpoint_QueryParams(t *testing.T) { } } -// TestGetVirtualKeys_FromMemoryUsesConfigStore verifies the legacy -// from_memory flag no longer bypasses the DB-backed ConfigStore path. -func TestGetVirtualKeys_FromMemoryUsesConfigStore(t *testing.T) { +// TestGetVirtualKeys_FromMemoryUsesGovernanceData verifies the from_memory +// flag serves virtual keys from the in-memory GovernanceData and bypasses the +// DB-backed ConfigStore entirely. +func TestGetVirtualKeys_FromMemoryUsesGovernanceData(t *testing.T) { SetLogger(&mockLogger{}) store := &mockConfigStoreForVK{} - manager := &mockGovernanceManagerForVK{} + manager := &mockGovernanceManagerForVK{ + data: &governance.GovernanceData{ + VirtualKeys: map[string]*configstoreTables.TableVirtualKey{}, + }, + } h := &GovernanceHandler{ configStore: store, governanceManager: manager, @@ -2101,24 +2223,29 @@ func TestGetVirtualKeys_FromMemoryUsesConfigStore(t *testing.T) { if ctx.Response.StatusCode() != 200 { t.Fatalf("expected status 200, got %d: %s", ctx.Response.StatusCode(), string(ctx.Response.Body())) } - if manager.getGovernanceDataCalls != 0 { - t.Fatalf("from_memory path called GetGovernanceData %d times", manager.getGovernanceDataCalls) + if manager.getGovernanceDataCalls != 1 { + t.Fatalf("expected GetGovernanceData to be called once, got %d", manager.getGovernanceDataCalls) } - if store.getVirtualKeysCalls != 1 { - t.Fatalf("expected GetVirtualKeys to be called once, got %d", store.getVirtualKeysCalls) + if store.getVirtualKeysCalls != 0 { + t.Fatalf("from_memory path called GetVirtualKeys %d times", store.getVirtualKeysCalls) } if store.getVirtualKeysPaginatedCalls != 0 { - t.Fatalf("unexpected paginated call count %d", store.getVirtualKeysPaginatedCalls) + t.Fatalf("from_memory path called GetVirtualKeysPaginated %d times", store.getVirtualKeysPaginatedCalls) } } -// TestGetVirtualKeys_FromMemoryWithLimitUsesPaginatedConfigStore verifies -// limit=0 plus from_memory still follows the DB-backed paginated path. -func TestGetVirtualKeys_FromMemoryWithLimitUsesPaginatedConfigStore(t *testing.T) { +// TestGetVirtualKeys_FromMemoryTakesPrecedenceOverLimit verifies the +// from_memory flag is honored even when pagination parameters are present, so +// the in-memory path is used and the paginated ConfigStore query is skipped. +func TestGetVirtualKeys_FromMemoryTakesPrecedenceOverLimit(t *testing.T) { SetLogger(&mockLogger{}) store := &mockConfigStoreForVK{} - manager := &mockGovernanceManagerForVK{} + manager := &mockGovernanceManagerForVK{ + data: &governance.GovernanceData{ + VirtualKeys: map[string]*configstoreTables.TableVirtualKey{}, + }, + } h := &GovernanceHandler{ configStore: store, governanceManager: manager, @@ -2133,14 +2260,14 @@ func TestGetVirtualKeys_FromMemoryWithLimitUsesPaginatedConfigStore(t *testing.T if ctx.Response.StatusCode() != 200 { t.Fatalf("expected status 200, got %d: %s", ctx.Response.StatusCode(), string(ctx.Response.Body())) } - if manager.getGovernanceDataCalls != 0 { - t.Fatalf("from_memory path called GetGovernanceData %d times", manager.getGovernanceDataCalls) + if manager.getGovernanceDataCalls != 1 { + t.Fatalf("expected GetGovernanceData to be called once, got %d", manager.getGovernanceDataCalls) } - if store.getVirtualKeysPaginatedCalls != 1 { - t.Fatalf("expected GetVirtualKeysPaginated to be called once, got %d", store.getVirtualKeysPaginatedCalls) + if store.getVirtualKeysPaginatedCalls != 0 { + t.Fatalf("from_memory path called GetVirtualKeysPaginated %d times", store.getVirtualKeysPaginatedCalls) } if store.getVirtualKeysCalls != 0 { - t.Fatalf("unexpected non-paginated call count %d", store.getVirtualKeysCalls) + t.Fatalf("from_memory path called GetVirtualKeys %d times", store.getVirtualKeysCalls) } } @@ -2875,3 +3002,235 @@ func TestApplyVKGovernanceFromModelConfigs_OverlaysModelConfigGovernance(t *test t.Errorf("expected model-config rate limit overlaid, got rl=%v id=%v", vk.RateLimit, vk.RateLimitID) } } + +// newGovernanceProviderNameCtx builds a RequestCtx exactly as the fasthttp router +// would hand it to the handler: the {provider_name} path param is stored RAW +// (still percent-encoded), because the router does not decode path params. This +// is what exercises url.PathUnescape inside the handler. +func newGovernanceProviderNameCtx(encodedProviderName, body string) *fasthttp.RequestCtx { + ctx := newTestRequestCtx(body) + ctx.SetUserValue("provider_name", encodedProviderName) + return ctx +} + +// TestProviderGovernance_DecodesEncodedProviderName is a regression test for the +// 404 "Provider not found" that occurred when updating/deleting governance for a +// custom provider whose name contains a space (e.g. "OpenRouter Base"). The UI +// percent-encodes the name in the path ("OpenRouter%20Base"); the handler must +// url.PathUnescape it before matching against the stored provider name. +func TestProviderGovernance_DecodesEncodedProviderName(t *testing.T) { + SetLogger(&mockLogger{}) + ctx := context.Background() + store := setupPricingOverrideHandlerStore(t) + handler := &GovernanceHandler{ + configStore: store, + governanceManager: pricingOverrideTestGovernanceManager{}, + } + + // Seed a custom provider whose name contains a space. + const providerName = "OpenRouter Base" + const encodedName = "OpenRouter%20Base" + if err := store.AddProvider(ctx, schemas.ModelProvider(providerName), configstore.ProviderConfig{}); err != nil { + t.Fatalf("seed provider: %v", err) + } + + // PUT a budget using the encoded name in the path param, exactly as the router + // delivers it. Before the fix this returned 404 because "OpenRouter%20Base" was + // compared raw against the stored name. + putCtx := newGovernanceProviderNameCtx(encodedName, `{"budgets":[{"max_limit":10,"reset_duration":"1M"}],"calendar_aligned":false}`) + handler.updateProviderGovernance(putCtx) + if putCtx.Response.StatusCode() != fasthttp.StatusOK { + t.Fatalf("PUT status got %d, want 200; body=%s", putCtx.Response.StatusCode(), putCtx.Response.Body()) + } + + // The budget must be persisted against the decoded provider name. + pn := providerName + mc, err := store.GetModelConfig(ctx, configstoreTables.ModelConfigScopeGlobal, nil, configstoreTables.ModelConfigAllModels, &pn) + if err != nil { + t.Fatalf("expected persisted model config for %q, got err: %v", providerName, err) + } + if len(mc.Budgets) != 1 || mc.Budgets[0].MaxLimit != 10 { + t.Fatalf("expected one budget with max_limit 10, got %+v", mc.Budgets) + } + + // DELETE with the same encoded path param must also resolve and succeed. + delCtx := newGovernanceProviderNameCtx(encodedName, "") + handler.deleteProviderGovernance(delCtx) + if delCtx.Response.StatusCode() != fasthttp.StatusOK { + t.Fatalf("DELETE status got %d, want 200; body=%s", delCtx.Response.StatusCode(), delCtx.Response.Body()) + } + + // The model config must actually be gone — a 200 alone could come from the + // handler's idempotent ErrNotFound branch even if nothing was removed. + if _, err := store.GetModelConfig(ctx, configstoreTables.ModelConfigScopeGlobal, nil, configstoreTables.ModelConfigAllModels, &pn); !errors.Is(err, configstore.ErrNotFound) { + t.Fatalf("expected model config for %q to be removed (ErrNotFound), got err: %v", providerName, err) + } +} + +// TestProviderGovernance_UnknownProviderStill404 guards the inverse: a genuinely +// unknown provider must still 404, so the decode change didn't mask the check. +func TestProviderGovernance_UnknownProviderStill404(t *testing.T) { + SetLogger(&mockLogger{}) + store := setupPricingOverrideHandlerStore(t) + handler := &GovernanceHandler{ + configStore: store, + governanceManager: pricingOverrideTestGovernanceManager{}, + } + + putCtx := newGovernanceProviderNameCtx("Nope%20Missing", `{"budgets":[{"max_limit":10,"reset_duration":"1M"}]}`) + handler.updateProviderGovernance(putCtx) + if putCtx.Response.StatusCode() != fasthttp.StatusNotFound { + t.Fatalf("PUT unknown provider status got %d, want 404; body=%s", putCtx.Response.StatusCode(), putCtx.Response.Body()) + } +} + +// TestProviderGovernance_MalformedEncodingReturns400 locks in the fail-closed +// contract: when the provider name is not valid percent-encoding (e.g. a stray +// "%2"), url.PathUnescape fails and both handlers must respond 400 rather than +// matching against the raw string. +func TestProviderGovernance_MalformedEncodingReturns400(t *testing.T) { + SetLogger(&mockLogger{}) + store := setupPricingOverrideHandlerStore(t) + handler := &GovernanceHandler{ + configStore: store, + governanceManager: pricingOverrideTestGovernanceManager{}, + } + + const malformedName = "OpenRouter%2" + + putCtx := newGovernanceProviderNameCtx(malformedName, `{"budgets":[{"max_limit":10,"reset_duration":"1M"}]}`) + handler.updateProviderGovernance(putCtx) + if putCtx.Response.StatusCode() != fasthttp.StatusBadRequest { + t.Fatalf("PUT malformed encoding status got %d, want 400; body=%s", putCtx.Response.StatusCode(), putCtx.Response.Body()) + } + + delCtx := newGovernanceProviderNameCtx(malformedName, "") + handler.deleteProviderGovernance(delCtx) + if delCtx.Response.StatusCode() != fasthttp.StatusBadRequest { + t.Fatalf("DELETE malformed encoding status got %d, want 400; body=%s", delCtx.Response.StatusCode(), delCtx.Response.Body()) + } +} + +// newGovernanceTeamIDCtx builds a request whose team_id path param carries the +// raw (still percent-encoded) value, exactly as the fasthttp router delivers it +// (it matches on URI().PathOriginal(), so no decoding happens before the handler). +func newGovernanceTeamIDCtx(encodedTeamID, body string) *fasthttp.RequestCtx { + ctx := newTestRequestCtx(body) + ctx.SetUserValue("team_id", encodedTeamID) + return ctx +} + +// TestTeam_DecodesEncodedTeamID is a regression test for #3106: SCIM/IdP-synced +// team IDs containing spaces or other URL-sensitive characters are listable but +// individual GET/DELETE returned "404 Team not found". The router delivers the +// team_id path segment still percent-encoded, so the handler must url.PathUnescape +// it before the config-store lookup (mirrors the provider_name handling). +func TestTeam_DecodesEncodedTeamID(t *testing.T) { + SetLogger(&mockLogger{}) + ctx := context.Background() + store := setupPricingOverrideHandlerStore(t) + handler := &GovernanceHandler{ + configStore: store, + governanceManager: pricingOverrideTestGovernanceManager{}, + } + + cases := []struct { + name string + teamID string // stored (decoded) ID + encoded string // what the router hands to the handler + }{ + {"space", "SCIM Team Alpha", "SCIM%20Team%20Alpha"}, + {"punctuation", "Team (prod): eu-west", "Team%20%28prod%29%3A%20eu-west"}, + {"encoded-slash", "org/team/beta", "org%2Fteam%2Fbeta"}, + {"plus", "team+gamma", "team%2Bgamma"}, + {"unicode", "команда-δ", "%D0%BA%D0%BE%D0%BC%D0%B0%D0%BD%D0%B4%D0%B0-%CE%B4"}, + } + + for i, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + team := &configstoreTables.TableTeam{ + ID: tc.teamID, + Name: tc.name + "-" + string(rune('A'+i)), // Name has a unique index + } + if err := store.CreateTeam(ctx, team); err != nil { + t.Fatalf("seed team %q: %v", tc.teamID, err) + } + + // GET with the encoded path param must resolve the team. + getCtx := newGovernanceTeamIDCtx(tc.encoded, "") + handler.getTeam(getCtx) + if getCtx.Response.StatusCode() != fasthttp.StatusOK { + t.Fatalf("GET status got %d, want 200; body=%s", getCtx.Response.StatusCode(), getCtx.Response.Body()) + } + var getResp struct { + Team configstoreTables.TableTeam `json:"team"` + } + if err := json.Unmarshal(getCtx.Response.Body(), &getResp); err != nil { + t.Fatalf("parse GET body: %v", err) + } + if getResp.Team.ID != tc.teamID { + t.Fatalf("GET returned team id %q, want %q", getResp.Team.ID, tc.teamID) + } + + // PUT with the encoded path param must resolve and apply the update. + renamed := tc.name + "-renamed-" + string(rune('A'+i)) + putCtx := newGovernanceTeamIDCtx(tc.encoded, `{"name":"`+renamed+`"}`) + handler.updateTeam(putCtx) + if putCtx.Response.StatusCode() != fasthttp.StatusOK { + t.Fatalf("PUT status got %d, want 200; body=%s", putCtx.Response.StatusCode(), putCtx.Response.Body()) + } + updated, err := store.GetTeam(ctx, tc.teamID) + if err != nil { + t.Fatalf("re-fetch team %q after update: %v", tc.teamID, err) + } + if updated.Name != renamed { + t.Fatalf("PUT did not apply: name %q, want %q", updated.Name, renamed) + } + + // DELETE with the same encoded path param must resolve and succeed. + delCtx := newGovernanceTeamIDCtx(tc.encoded, "") + handler.deleteTeam(delCtx) + if delCtx.Response.StatusCode() != fasthttp.StatusOK { + t.Fatalf("DELETE status got %d, want 200; body=%s", delCtx.Response.StatusCode(), delCtx.Response.Body()) + } + + // The team must actually be gone — a 200 could otherwise come from an + // idempotent ErrNotFound branch without deleting anything. + if _, err := store.GetTeam(ctx, tc.teamID); !errors.Is(err, configstore.ErrNotFound) { + t.Fatalf("expected team %q removed (ErrNotFound), got err: %v", tc.teamID, err) + } + }) + } +} + +// TestTeam_MalformedEncodingReturns400 locks in the fail-closed contract: a team_id +// that is not valid percent-encoding (e.g. a stray "%2") must yield 400 rather than +// being matched raw against stored IDs. +func TestTeam_MalformedEncodingReturns400(t *testing.T) { + SetLogger(&mockLogger{}) + store := setupPricingOverrideHandlerStore(t) + handler := &GovernanceHandler{ + configStore: store, + governanceManager: pricingOverrideTestGovernanceManager{}, + } + + const malformedID = "Team%2" + + getCtx := newGovernanceTeamIDCtx(malformedID, "") + handler.getTeam(getCtx) + if getCtx.Response.StatusCode() != fasthttp.StatusBadRequest { + t.Fatalf("GET malformed encoding status got %d, want 400; body=%s", getCtx.Response.StatusCode(), getCtx.Response.Body()) + } + + putCtx := newGovernanceTeamIDCtx(malformedID, `{"name":"irrelevant"}`) + handler.updateTeam(putCtx) + if putCtx.Response.StatusCode() != fasthttp.StatusBadRequest { + t.Fatalf("PUT malformed encoding status got %d, want 400; body=%s", putCtx.Response.StatusCode(), putCtx.Response.Body()) + } + + delCtx := newGovernanceTeamIDCtx(malformedID, "") + handler.deleteTeam(delCtx) + if delCtx.Response.StatusCode() != fasthttp.StatusBadRequest { + t.Fatalf("DELETE malformed encoding status got %d, want 400; body=%s", delCtx.Response.StatusCode(), delCtx.Response.Body()) + } +} diff --git a/transports/bifrost-http/handlers/inference.go b/transports/bifrost-http/handlers/inference.go index f98726dbe89..6ccd892d54c 100644 --- a/transports/bifrost-http/handlers/inference.go +++ b/transports/bifrost-http/handlers/inference.go @@ -4,7 +4,6 @@ package handlers import ( "context" - "encoding/base64" "encoding/json" "errors" @@ -25,6 +24,7 @@ import ( providerUtils "github.com/maximhq/bifrost/core/providers/utils" "github.com/maximhq/bifrost/core/schemas" + "github.com/maximhq/bifrost/framework/modelcatalog" "github.com/maximhq/bifrost/transports/bifrost-http/lib" "github.com/valyala/fasthttp" ) @@ -75,7 +75,7 @@ func resolveModelAndProvider(_ *fasthttp.RequestCtx, _ *lib.Config, model string func prepareRequest[T baseRequest](ctx *fasthttp.RequestCtx, config *lib.Config, knownFields map[string]bool) (*T, *requestBase, error) { req := new(T) if err := sonic.Unmarshal(ctx.PostBody(), req); err != nil { - return nil, nil, fmt.Errorf("invalid request format: %v", err) + return nil, nil, fmt.Errorf("Invalid request payload") } provider, modelName, err := resolveModelAndProvider(ctx, config, (*req).getModel()) if err != nil { @@ -725,6 +725,14 @@ func (h *CompletionHandler) RegisterRoutes(r *router.Router, middlewares ...sche r.POST("/v1/completions", lib.ChainMiddlewares(h.textCompletion, baseMiddlewares...)) r.POST("/v1/chat/completions", lib.ChainMiddlewares(h.chatCompletion, baseMiddlewares...)) r.POST("/v1/responses", lib.ChainMiddlewares(h.responses, baseMiddlewares...)) + responsesRetrieveMW := append([]schemas.BifrostHTTPMiddleware{createRequestTypeMiddleware(schemas.ResponsesRetrieveRequest)}, middlewares...) + responsesDeleteMW := append([]schemas.BifrostHTTPMiddleware{createRequestTypeMiddleware(schemas.ResponsesDeleteRequest)}, middlewares...) + responsesCancelMW := append([]schemas.BifrostHTTPMiddleware{createRequestTypeMiddleware(schemas.ResponsesCancelRequest)}, middlewares...) + responsesInputItemsMW := append([]schemas.BifrostHTTPMiddleware{createRequestTypeMiddleware(schemas.ResponsesInputItemsRequest)}, middlewares...) + r.GET("/v1/responses/{response_id}", lib.ChainMiddlewares(h.responsesRetrieve, responsesRetrieveMW...)) + r.DELETE("/v1/responses/{response_id}", lib.ChainMiddlewares(h.responsesDelete, responsesDeleteMW...)) + r.POST("/v1/responses/{response_id}/cancel", lib.ChainMiddlewares(h.responsesCancel, responsesCancelMW...)) + r.GET("/v1/responses/{response_id}/input_items", lib.ChainMiddlewares(h.responsesInputItems, responsesInputItemsMW...)) r.POST("/v1/embeddings", lib.ChainMiddlewares(h.embeddings, baseMiddlewares...)) r.POST("/v1/rerank", lib.ChainMiddlewares(h.rerank, baseMiddlewares...)) r.POST("/v1/ocr", lib.ChainMiddlewares(h.ocr, baseMiddlewares...)) @@ -863,67 +871,77 @@ func (h *CompletionHandler) listModels(ctx *fasthttp.RequestCtx) { return } - // Add pricing data to the response - if len(resp.Data) > 0 && h.config.ModelCatalog != nil { - for i, modelEntry := range resp.Data { - provider, modelName := schemas.ParseModelString(modelEntry.ID, "") - pricingEntry := h.config.ModelCatalog.GetPricingEntryForModel(modelName, provider) - if pricingEntry == nil && modelEntry.Alias != nil { - // Retry with alias - pricingEntry = h.config.ModelCatalog.GetPricingEntryForModel(*modelEntry.Alias, provider) + enrichListModelsResponse(resp, h.config.ModelCatalog) + if resp != nil && resp.ExtraFields.ProviderResponseHeaders != nil { + forwardProviderHeaders(ctx, resp.ExtraFields.ProviderResponseHeaders) + } + // Send successful response + SendJSON(ctx, resp) +} + +func enrichListModelsResponse(resp *schemas.BifrostListModelsResponse, catalog *modelcatalog.ModelCatalog) { + if resp == nil || len(resp.Data) == 0 { + return + } + + if catalog == nil { + return + } + + for i := range resp.Data { + modelEntry := resp.Data[i] + provider, modelName := schemas.ParseModelString(modelEntry.ID, "") + pricingEntry := catalog.GetPricingEntryForModel(modelName, provider) + if pricingEntry == nil && modelEntry.Alias != nil { + pricingEntry = catalog.GetPricingEntryForModel(*modelEntry.Alias, provider) + } + if pricingEntry != nil { + modelEntry.IsDeprecated = modelEntry.IsDeprecated || pricingEntry.IsDeprecated + if pricingEntry.BaseModel != "" && modelEntry.NormalizedName == nil { + modelEntry.NormalizedName = bifrost.Ptr(providerUtils.NormalizeBaseModelSlug(pricingEntry.BaseModel)) } - if pricingEntry != nil { - if pricingEntry.BaseModel != "" && resp.Data[i].NormalizedName == nil { - resp.Data[i].NormalizedName = bifrost.Ptr(providerUtils.NormalizeBaseModelSlug(pricingEntry.BaseModel)) - } - if len(pricingEntry.AdditionalAttributes) > 0 && resp.Data[i].AdditionalAttributes == nil { - resp.Data[i].AdditionalAttributes = pricingEntry.AdditionalAttributes + if len(pricingEntry.AdditionalAttributes) > 0 && modelEntry.AdditionalAttributes == nil { + modelEntry.AdditionalAttributes = pricingEntry.AdditionalAttributes + } + if pricingEntry.ContextLength != nil && modelEntry.ContextLength == nil { + modelEntry.ContextLength = pricingEntry.ContextLength + } else if pricingEntry.MaxInputTokens != nil && modelEntry.ContextLength == nil { + modelEntry.ContextLength = pricingEntry.MaxInputTokens + } + if pricingEntry.MaxInputTokens != nil && modelEntry.MaxInputTokens == nil { + modelEntry.MaxInputTokens = pricingEntry.MaxInputTokens + } + if pricingEntry.MaxOutputTokens != nil && modelEntry.MaxOutputTokens == nil { + modelEntry.MaxOutputTokens = pricingEntry.MaxOutputTokens + } + if pricingEntry.Architecture != nil && modelEntry.Architecture == nil { + modelEntry.Architecture = pricingEntry.Architecture + } + if modelEntry.Pricing == nil { + pricing := &schemas.Pricing{} + if pricingEntry.InputCostPerToken != nil { + pricing.Prompt = bifrost.Ptr(fmt.Sprintf("%.10f", *pricingEntry.InputCostPerToken)) } - if pricingEntry.ContextLength != nil && resp.Data[i].ContextLength == nil { - resp.Data[i].ContextLength = pricingEntry.ContextLength - } else if pricingEntry.MaxInputTokens != nil && resp.Data[i].ContextLength == nil { - resp.Data[i].ContextLength = pricingEntry.MaxInputTokens // fallback to MaxInputTokens if ContextLength is not set + if pricingEntry.OutputCostPerToken != nil { + pricing.Completion = bifrost.Ptr(fmt.Sprintf("%.10f", *pricingEntry.OutputCostPerToken)) } - - if pricingEntry.MaxInputTokens != nil && resp.Data[i].MaxInputTokens == nil { - resp.Data[i].MaxInputTokens = pricingEntry.MaxInputTokens + if pricingEntry.InputCostPerImage != nil { + pricing.Image = bifrost.Ptr(fmt.Sprintf("%.10f", *pricingEntry.InputCostPerImage)) } - if pricingEntry.MaxOutputTokens != nil && resp.Data[i].MaxOutputTokens == nil { - resp.Data[i].MaxOutputTokens = pricingEntry.MaxOutputTokens + if pricingEntry.CacheReadInputTokenCost != nil { + pricing.InputCacheRead = bifrost.Ptr(fmt.Sprintf("%.10f", *pricingEntry.CacheReadInputTokenCost)) } - if pricingEntry.Architecture != nil && resp.Data[i].Architecture == nil { - resp.Data[i].Architecture = pricingEntry.Architecture + if pricingEntry.CacheCreationInputTokenCost != nil { + pricing.InputCacheWrite = bifrost.Ptr(fmt.Sprintf("%.10f", *pricingEntry.CacheCreationInputTokenCost)) } - if modelEntry.Pricing == nil { - pricing := &schemas.Pricing{} - if pricingEntry.InputCostPerToken != nil { - pricing.Prompt = bifrost.Ptr(fmt.Sprintf("%.10f", *pricingEntry.InputCostPerToken)) - } - if pricingEntry.OutputCostPerToken != nil { - pricing.Completion = bifrost.Ptr(fmt.Sprintf("%.10f", *pricingEntry.OutputCostPerToken)) - } - if pricingEntry.InputCostPerImage != nil { - pricing.Image = bifrost.Ptr(fmt.Sprintf("%.10f", *pricingEntry.InputCostPerImage)) - } - if pricingEntry.CacheReadInputTokenCost != nil { - pricing.InputCacheRead = bifrost.Ptr(fmt.Sprintf("%.10f", *pricingEntry.CacheReadInputTokenCost)) - } - if pricingEntry.CacheCreationInputTokenCost != nil { - pricing.InputCacheWrite = bifrost.Ptr(fmt.Sprintf("%.10f", *pricingEntry.CacheCreationInputTokenCost)) - } - if pricingEntry.SearchContextCostPerQuery != nil { - pricing.WebSearch = bifrost.Ptr(fmt.Sprintf("%.10f", *pricingEntry.SearchContextCostPerQuery)) - } - resp.Data[i].Pricing = pricing + if pricingEntry.SearchContextCostPerQuery != nil { + pricing.WebSearch = bifrost.Ptr(fmt.Sprintf("%.10f", *pricingEntry.SearchContextCostPerQuery)) } + modelEntry.Pricing = pricing } } + resp.Data[i] = modelEntry } - if resp != nil && resp.ExtraFields.ProviderResponseHeaders != nil { - forwardProviderHeaders(ctx, resp.ExtraFields.ProviderResponseHeaders) - } - // Send successful response - SendJSON(ctx, resp) } // prepareTextCompletionRequest prepares a BifrostTextCompletionRequest from the HTTP request body @@ -1542,6 +1560,14 @@ func (h *CompletionHandler) countTokens(ctx *fasthttp.RequestCtx) { SendJSON(ctx, response) } +func responsesLifecycleProviderFromQuery(ctx *fasthttp.RequestCtx) schemas.ModelProvider { + p := schemas.ModelProvider(string(ctx.QueryArgs().Peek("provider"))) + if p == "" { + return schemas.OpenAI + } + return p +} + // prepareCompactionRequest prepares a BifrostCompactionRequest from the HTTP request body func prepareCompactionRequest(ctx *fasthttp.RequestCtx, config *lib.Config) (*CompactionHTTPRequest, *schemas.BifrostCompactionRequest, error) { req, base, err := prepareRequest[CompactionHTTPRequest](ctx, config, compactionParamsKnownFields) @@ -1577,6 +1603,60 @@ func prepareCompactionRequest(ctx *fasthttp.RequestCtx, config *lib.Config) (*Co }, nil } +// responsesRetrieve handles GET /v1/responses/{response_id}. +func (h *CompletionHandler) responsesRetrieve(ctx *fasthttp.RequestCtx) { + responseID, ok := ctx.UserValue("response_id").(string) + if !ok || responseID == "" { + SendError(ctx, fasthttp.StatusBadRequest, "response_id is required") + return + } + bifrostReq := &schemas.BifrostResponsesRetrieveRequest{ + Provider: responsesLifecycleProviderFromQuery(ctx), + ResponseID: responseID, + } + ctx.QueryArgs().VisitAll(func(key, value []byte) { + switch string(key) { + case "include": + bifrostReq.Include = append(bifrostReq.Include, string(value)) + } + }) + if raw := ctx.QueryArgs().Peek("starting_after"); len(raw) > 0 { + n, err := strconv.Atoi(string(raw)) + if err != nil { + SendError(ctx, fasthttp.StatusBadRequest, "starting_after must be an integer") + return + } + bifrostReq.StartingAfter = schemas.Ptr(n) + } + if raw := ctx.QueryArgs().Peek("include_obfuscation"); len(raw) > 0 { + b, err := strconv.ParseBool(string(raw)) + if err != nil { + SendError(ctx, fasthttp.StatusBadRequest, "include_obfuscation must be a boolean") + return + } + bifrostReq.IncludeObfuscation = &b + } + bifrostCtx, cancel := lib.ConvertToBifrostContext(ctx, h.config) + defer cancel() + if bifrostCtx == nil { + SendError(ctx, fasthttp.StatusBadRequest, "Failed to convert context") + return + } + resp, bifrostErr := h.client.ResponsesRetrieveRequest(bifrostCtx, bifrostReq) + if bifrostErr != nil { + forwardProviderHeadersFromContext(ctx, bifrostCtx) + SendBifrostError(ctx, bifrostErr) + return + } + if resp != nil && resp.ExtraFields.ProviderResponseHeaders != nil { + forwardProviderHeaders(ctx, resp.ExtraFields.ProviderResponseHeaders) + } + if streamLargeResponseIfActive(ctx, bifrostCtx) { + return + } + SendJSON(ctx, resp) +} + // compaction handles POST /v1/responses/compact - Compact a conversation context window func (h *CompletionHandler) compaction(ctx *fasthttp.RequestCtx) { _, bifrostCompactionReq, err := prepareCompactionRequest(ctx, h.config) @@ -1586,11 +1666,11 @@ func (h *CompletionHandler) compaction(ctx *fasthttp.RequestCtx) { } bifrostCtx, cancel := lib.ConvertToBifrostContext(ctx, h.config) + defer cancel() if bifrostCtx == nil { SendError(ctx, fasthttp.StatusBadRequest, "Failed to convert context") return } - defer cancel() response, bifrostErr := h.client.CompactionRequest(bifrostCtx, bifrostCompactionReq) if bifrostErr != nil { @@ -1608,6 +1688,120 @@ func (h *CompletionHandler) compaction(ctx *fasthttp.RequestCtx) { SendJSON(ctx, response) } +// responsesDelete handles DELETE /v1/responses/{response_id}. +func (h *CompletionHandler) responsesDelete(ctx *fasthttp.RequestCtx) { + responseID, ok := ctx.UserValue("response_id").(string) + if !ok || responseID == "" { + SendError(ctx, fasthttp.StatusBadRequest, "response_id is required") + return + } + bifrostReq := &schemas.BifrostResponsesDeleteRequest{ + Provider: responsesLifecycleProviderFromQuery(ctx), + ResponseID: responseID, + } + bifrostCtx, cancel := lib.ConvertToBifrostContext(ctx, h.config) + defer cancel() + if bifrostCtx == nil { + SendError(ctx, fasthttp.StatusBadRequest, "Failed to convert context") + return + } + resp, bifrostErr := h.client.ResponsesDeleteRequest(bifrostCtx, bifrostReq) + if bifrostErr != nil { + forwardProviderHeadersFromContext(ctx, bifrostCtx) + SendBifrostError(ctx, bifrostErr) + return + } + if resp != nil && resp.ExtraFields.ProviderResponseHeaders != nil { + forwardProviderHeaders(ctx, resp.ExtraFields.ProviderResponseHeaders) + } + if streamLargeResponseIfActive(ctx, bifrostCtx) { + return + } + SendJSON(ctx, resp) +} + +// responsesCancel handles POST /v1/responses/{response_id}/cancel. +func (h *CompletionHandler) responsesCancel(ctx *fasthttp.RequestCtx) { + responseID, ok := ctx.UserValue("response_id").(string) + if !ok || responseID == "" { + SendError(ctx, fasthttp.StatusBadRequest, "response_id is required") + return + } + bifrostReq := &schemas.BifrostResponsesCancelRequest{ + Provider: responsesLifecycleProviderFromQuery(ctx), + ResponseID: responseID, + } + bifrostCtx, cancel := lib.ConvertToBifrostContext(ctx, h.config) + defer cancel() + if bifrostCtx == nil { + SendError(ctx, fasthttp.StatusBadRequest, "Failed to convert context") + return + } + resp, bifrostErr := h.client.ResponsesCancelRequest(bifrostCtx, bifrostReq) + if bifrostErr != nil { + forwardProviderHeadersFromContext(ctx, bifrostCtx) + SendBifrostError(ctx, bifrostErr) + return + } + if resp != nil && resp.ExtraFields.ProviderResponseHeaders != nil { + forwardProviderHeaders(ctx, resp.ExtraFields.ProviderResponseHeaders) + } + if streamLargeResponseIfActive(ctx, bifrostCtx) { + return + } + SendJSON(ctx, resp) +} + +// responsesInputItems handles GET /v1/responses/{response_id}/input_items. +func (h *CompletionHandler) responsesInputItems(ctx *fasthttp.RequestCtx) { + responseID, ok := ctx.UserValue("response_id").(string) + if !ok || responseID == "" { + SendError(ctx, fasthttp.StatusBadRequest, "response_id is required") + return + } + bifrostReq := &schemas.BifrostResponsesInputItemsRequest{ + Provider: responsesLifecycleProviderFromQuery(ctx), + ResponseID: responseID, + } + ctx.QueryArgs().VisitAll(func(key, value []byte) { + switch string(key) { + case "after": + bifrostReq.After = string(value) + case "include": + bifrostReq.Include = append(bifrostReq.Include, string(value)) + case "order": + bifrostReq.Order = string(value) + } + }) + if raw := ctx.QueryArgs().Peek("limit"); len(raw) > 0 { + n, err := strconv.Atoi(string(raw)) + if err != nil { + SendError(ctx, fasthttp.StatusBadRequest, "limit must be an integer") + return + } + bifrostReq.Limit = schemas.Ptr(n) + } + bifrostCtx, cancel := lib.ConvertToBifrostContext(ctx, h.config) + defer cancel() + if bifrostCtx == nil { + SendError(ctx, fasthttp.StatusBadRequest, "Failed to convert context") + return + } + resp, bifrostErr := h.client.ResponsesInputItemsRequest(bifrostCtx, bifrostReq) + if bifrostErr != nil { + forwardProviderHeadersFromContext(ctx, bifrostCtx) + SendBifrostError(ctx, bifrostErr) + return + } + if resp != nil && resp.ExtraFields.ProviderResponseHeaders != nil { + forwardProviderHeaders(ctx, resp.ExtraFields.ProviderResponseHeaders) + } + if streamLargeResponseIfActive(ctx, bifrostCtx) { + return + } + SendJSON(ctx, resp) +} + // handleStreamingTextCompletion handles streaming text completion requests using Server-Sent Events (SSE) func (h *CompletionHandler) handleStreamingTextCompletion(ctx *fasthttp.RequestCtx, req *schemas.BifrostTextCompletionRequest, bifrostCtx *schemas.BifrostContext, cancel context.CancelFunc) { // Use the cancellable context from ConvertToBifrostContext @@ -2379,7 +2573,7 @@ func (h *CompletionHandler) imageVariation(ctx *fasthttp.RequestCtx) { func (h *CompletionHandler) videoGeneration(ctx *fasthttp.RequestCtx) { var req VideoGenerationRequest if err := sonic.Unmarshal(ctx.PostBody(), &req); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid request format: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } @@ -2691,7 +2885,7 @@ func (h *CompletionHandler) videoRemix(ctx *fasthttp.RequestCtx) { // Parse request body var req VideoRemixRequest if err := sonic.Unmarshal(ctx.PostBody(), &req); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid request format: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } @@ -2767,7 +2961,7 @@ func resolveBatchProvider(ctx *fasthttp.RequestCtx, config *lib.Config, model st func (h *CompletionHandler) batchCreate(ctx *fasthttp.RequestCtx) { var req BatchCreateRequest if err := sonic.Unmarshal(ctx.PostBody(), &req); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid request format: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } @@ -3113,10 +3307,12 @@ func (h *CompletionHandler) fileUpload(ctx *fasthttp.RequestCtx) { // when omitted, the provider mints an upload session URL instead of receiving bytes) var fileData []byte var filename string + var filePartContentType string fileHeaders := form.File["file"] if len(fileHeaders) > 0 { fileHeader := fileHeaders[0] filename = fileHeader.Filename + filePartContentType = strings.TrimSpace(fileHeader.Header.Get("Content-Type")) // Open and read the file file, err := fileHeader.Open() @@ -3142,6 +3338,8 @@ func (h *CompletionHandler) fileUpload(ctx *fasthttp.RequestCtx) { var contentType *string if len(form.Value["content_type"]) > 0 && form.Value["content_type"][0] != "" { contentType = &form.Value["content_type"][0] + } else if filePartContentType != "" { + contentType = &filePartContentType } // GCS storage location for Vertex uploads: sent as individual multipart fields, @@ -3463,7 +3661,7 @@ func (h *CompletionHandler) fileContent(ctx *fasthttp.RequestCtx) { func (h *CompletionHandler) containerCreate(ctx *fasthttp.RequestCtx) { var req ContainerCreateRequest if err := sonic.Unmarshal(ctx.PostBody(), &req); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid request format: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } @@ -3993,4 +4191,4 @@ func (h *CompletionHandler) containerFileDelete(ctx *fasthttp.RequestCtx) { return } SendJSON(ctx, resp) -} +} \ No newline at end of file diff --git a/transports/bifrost-http/handlers/list_models_vk_test.go b/transports/bifrost-http/handlers/list_models_vk_test.go index 59e18eec72b..890dae78aa2 100644 --- a/transports/bifrost-http/handlers/list_models_vk_test.go +++ b/transports/bifrost-http/handlers/list_models_vk_test.go @@ -28,7 +28,7 @@ func TestApplyListModelsVirtualKeyProviderFilterSetsActiveVKProviders(t *testing h := &CompletionHandler{ config: &lib.Config{ ConfigStore: &mockListModelsVKConfigStore{vk: &configstoreTables.TableVirtualKey{ - Value: "sk-bf-active", + Value: *schemas.NewSecretVar("sk-bf-active"), IsActive: new(true), ProviderConfigs: []configstoreTables.TableVirtualKeyProviderConfig{ {Provider: "openai"}, @@ -124,7 +124,7 @@ func TestApplyListModelsVirtualKeyProviderFilterSkipsInactiveVK(t *testing.T) { h := &CompletionHandler{ config: &lib.Config{ ConfigStore: &mockListModelsVKConfigStore{vk: &configstoreTables.TableVirtualKey{ - Value: "sk-bf-inactive", + Value: *schemas.NewSecretVar("sk-bf-inactive"), IsActive: new(false), ProviderConfigs: []configstoreTables.TableVirtualKeyProviderConfig{ {Provider: "openai"}, diff --git a/transports/bifrost-http/handlers/localhostcheck_test.go b/transports/bifrost-http/handlers/localhostcheck_test.go new file mode 100644 index 00000000000..de63ea268b2 --- /dev/null +++ b/transports/bifrost-http/handlers/localhostcheck_test.go @@ -0,0 +1,78 @@ +package handlers + +import "testing" + +func TestIsLocalhost(t *testing.T) { + tests := []struct { + host string + want bool + }{ + {"localhost", true}, + {"localhost:8080", true}, + {"127.0.0.1", true}, + {"127.0.0.1:8080", true}, + {"::1", true}, + {"[::1]", true}, + {"[::1]:8080", true}, + {"::ffff:127.0.0.1", true}, + {"", false}, + {"[::1", false}, + {"::1]", false}, + {"[localhost]", false}, + {"example.com", false}, + {"example.com:8080", false}, + {"evil-localhost.com:80", false}, + {"192.168.1.10:8080", false}, + {"[2001:db8::1]:8080", false}, + {"2001:db8::1", false}, + } + for _, tt := range tests { + if got := isLocalhost(tt.host); got != tt.want { + t.Errorf("isLocalhost(%q) = %v, want %v", tt.host, got, tt.want) + } + } +} + +func TestIsLocalhostOrigin(t *testing.T) { + tests := []struct { + origin string + want bool + }{ + {"http://localhost:3000", true}, + {"https://localhost:3000", true}, + {"http://127.0.0.1:3000", true}, + {"https://127.0.0.1:3000", true}, + {"http://0.0.0.0:3000", true}, + {"http://[::1]:3000", true}, + {"https://[::1]:3000", true}, + {"http://[::]:3000", true}, + {"http://localhost", true}, + {"http://example.com:3000", false}, + {"https://evil.com", false}, + {"http://[2001:db8::1]:3000", false}, + {"ftp://localhost:3000", false}, + {"not-a-url", false}, + {"", false}, + } + for _, tt := range tests { + if got := isLocalhostOrigin(tt.origin); got != tt.want { + t.Errorf("isLocalhostOrigin(%q) = %v, want %v", tt.origin, got, tt.want) + } + } +} + +func TestLoopbackRedirectURIsIPv6(t *testing.T) { + if !isAllowedRedirectScheme("http://[::1]:49152/cb") { + t.Error("http://[::1]:49152/cb should be an allowed redirect scheme (RFC 8252 §7.3)") + } + if isAllowedRedirectScheme("http://example.com/cb") { + t.Error("http on a non-loopback host must not be allowed") + } + // Loopback matching ignores the port and matches across loopback hosts + if !matchRedirectURI("http://[::1]:49152/cb", []string{"http://127.0.0.1:1234/cb"}) { + t.Error("IPv6 loopback redirect should match a registered IPv4 loopback URI (port-agnostic)") + } + if matchRedirectURI("http://[2001:db8::1]:49152/cb", []string{"http://[2001:db8::1]:1234/cb"}) { + t.Error("non-loopback IPv6 must require an exact match") + } +} diff --git a/transports/bifrost-http/handlers/logging.go b/transports/bifrost-http/handlers/logging.go index 8dd505e83cf..484cbc41399 100644 --- a/transports/bifrost-http/handlers/logging.go +++ b/transports/bifrost-http/handlers/logging.go @@ -1681,9 +1681,20 @@ func (h *LoggingHandler) recalculateLogCosts(ctx *fasthttp.RequestCtx) { limit = 1000 } - filters := payload.Filters + filters := payload.Filters.SearchFilters + if payload.Filters.Period != "" { + if start, end := ResolvePeriod(payload.Filters.Period); start != nil { + filters.StartTime = start + filters.EndTime = end + } + } filters.MissingCostOnly = true + if strings.Contains(string(ctx.Request.Header.Peek("Accept")), "text/event-stream") { + h.streamRecalculateLogCosts(ctx, filters, limit) + return + } + result, err := h.logManager.RecalculateCosts(ctx, &filters, limit) if err != nil { logger.Error("failed to recalculate log costs: %v", err) @@ -1694,6 +1705,50 @@ func (h *LoggingHandler) recalculateLogCosts(ctx *fasthttp.RequestCtx) { SendJSON(ctx, result) } +func (h *LoggingHandler) streamRecalculateLogCosts(ctx *fasthttp.RequestCtx, filters logstore.SearchFilters, limit int) { + ctx.SetContentType("text/event-stream") + ctx.Response.Header.Set("Cache-Control", "no-cache") + ctx.Response.Header.Set("Connection", "keep-alive") + ctx.Response.Header.Set("X-Accel-Buffering", "no") + + reader := lib.NewSSEStreamReader() + ctx.Response.SetBodyStream(reader, -1) + + streamCtx, cancel := context.WithCancel(ctx) + go func() { + defer reader.Done() + defer cancel() + + result, err := h.logManager.RecalculateCostsWithProgress(streamCtx, &filters, limit, func(progress logging.RecalculateCostProgress) { + data, marshalErr := sonic.Marshal(progress) + if marshalErr != nil { + logger.Warn("failed to marshal recalculate cost progress: %v", marshalErr) + return + } + if !reader.SendEvent("progress", data) { + cancel() + } + }) + if err != nil { + logger.Error("failed to recalculate log costs: %v", err) + data, marshalErr := sonic.Marshal(map[string]string{"message": fmt.Sprintf("Failed to recalculate costs: %v", err)}) + if marshalErr != nil { + data = []byte(`{"message":"Failed to recalculate costs"}`) + } + reader.SendError(data) + return + } + + data, marshalErr := sonic.Marshal(result) + if marshalErr != nil { + logger.Warn("failed to marshal recalculate cost result: %v", marshalErr) + reader.SendError([]byte(`{"message":"Failed to encode response"}`)) + return + } + reader.SendEvent("done", data) + }() +} + // Helper functions func findRedactedKey(redactedKeys []schemas.Key, id string, name string) *schemas.Key { @@ -1823,10 +1878,15 @@ func parseMetadataFilters(ctx *fasthttp.RequestCtx, filters *logstore.SearchFilt } type recalculateCostRequest struct { - Filters logstore.SearchFilters `json:"filters"` + Filters recalculateCostFilters `json:"filters"` Limit *int `json:"limit,omitempty"` } +type recalculateCostFilters struct { + logstore.SearchFilters + Period string `json:"period,omitempty"` +} + // parseMCPFiltersAndPagination parses MCP tool log filters and pagination from query parameters. // Returns an error if any required parsing fails (e.g., invalid time format, invalid number format). func parseMCPFiltersAndPagination(ctx *fasthttp.RequestCtx) (*logstore.MCPToolLogSearchFilters, *logstore.PaginationOptions, error) { diff --git a/transports/bifrost-http/handlers/logging_test.go b/transports/bifrost-http/handlers/logging_test.go index 63bca9d0f74..8e1eaa05bd7 100644 --- a/transports/bifrost-http/handlers/logging_test.go +++ b/transports/bifrost-http/handlers/logging_test.go @@ -6,6 +6,7 @@ import ( "errors" "net" "testing" + "time" "github.com/maximhq/bifrost/framework/logstore" "github.com/maximhq/bifrost/framework/queryscope" @@ -148,10 +149,73 @@ func TestGetDashboard(t *testing.T) { } } +func TestRecalculateLogCostsResolvesPeriodFilter(t *testing.T) { + SetLogger(&mockLogger{}) + + mgr := &dashboardLogManager{} + h := &LoggingHandler{logManager: mgr} + + var req fasthttp.Request + req.Header.SetMethod(fasthttp.MethodPost) + req.SetRequestURI("/api/logs/recalculate-cost") + req.Header.SetContentType("application/json") + req.SetBodyString(`{"filters":{"period":"1h"}}`) + + ctx := &fasthttp.RequestCtx{} + ctx.Init(&req, &net.TCPAddr{IP: net.IPv4(127, 0, 0, 1), Port: 12345}, nil) + + h.recalculateLogCosts(ctx) + + if got := ctx.Response.StatusCode(); got != fasthttp.StatusOK { + t.Fatalf("expected status 200, got %d: %s", got, string(ctx.Response.Body())) + } + filters := mgr.lastRecalculateFilters + if filters.StartTime == nil || filters.EndTime == nil { + t.Fatalf("expected period to resolve start/end, got start=%v end=%v", filters.StartTime, filters.EndTime) + } + if !filters.EndTime.After(*filters.StartTime) { + t.Fatalf("expected end_time after start_time, got start=%s end=%s", filters.StartTime, filters.EndTime) + } + if !filters.MissingCostOnly { + t.Fatal("expected recalculation to force missing_cost_only") + } +} + +func TestStreamRecalculateLogCostsUsesRequestContext(t *testing.T) { + SetLogger(&mockLogger{}) + + mgr := &dashboardLogManager{lastRecalculateContext: make(chan context.Context, 1)} + h := &LoggingHandler{logManager: mgr} + + var req fasthttp.Request + req.Header.SetMethod(fasthttp.MethodPost) + req.SetRequestURI("/api/logs/recalculate-cost") + req.Header.SetContentType("application/json") + req.Header.Set("Accept", "text/event-stream") + req.SetBodyString(`{"filters":{"period":"1h"}}`) + + ctx := &fasthttp.RequestCtx{} + ctx.Init(&req, &net.TCPAddr{IP: net.IPv4(127, 0, 0, 1), Port: 12345}, nil) + ctx.SetUserValue("request-scope", "preserved") + + h.recalculateLogCosts(ctx) + + select { + case recalculateCtx := <-mgr.lastRecalculateContext: + if got := recalculateCtx.Value("request-scope"); got != "preserved" { + t.Fatalf("expected request context value to be preserved, got %v", got) + } + case <-time.After(time.Second): + t.Fatal("timed out waiting for recalculation context") + } +} + type dashboardLogManager struct { - failStats bool - lastLLMFilters logstore.SearchFilters - lastMCPFilters logstore.MCPToolLogSearchFilters + failStats bool + lastLLMFilters logstore.SearchFilters + lastMCPFilters logstore.MCPToolLogSearchFilters + lastRecalculateFilters logstore.SearchFilters + lastRecalculateContext chan context.Context } func (m *dashboardLogManager) GetLog(ctx context.Context, id string) (*logstore.Log, error) { @@ -252,6 +316,14 @@ func (m *dashboardLogManager) GetDimensionLatencyHistogram(ctx context.Context, func (m *dashboardLogManager) DeleteLog(ctx context.Context, id string) error { return nil } func (m *dashboardLogManager) DeleteLogs(ctx context.Context, ids []string) error { return nil } func (m *dashboardLogManager) RecalculateCosts(ctx context.Context, filters *logstore.SearchFilters, limit int) (*loggingplugin.RecalculateCostResult, error) { + m.lastRecalculateFilters = *filters + return &loggingplugin.RecalculateCostResult{}, nil +} +func (m *dashboardLogManager) RecalculateCostsWithProgress(ctx context.Context, filters *logstore.SearchFilters, limit int, progress func(loggingplugin.RecalculateCostProgress)) (*loggingplugin.RecalculateCostResult, error) { + m.lastRecalculateFilters = *filters + if m.lastRecalculateContext != nil { + m.lastRecalculateContext <- ctx + } return nil, nil } func (m *dashboardLogManager) GetMCPToolLog(ctx context.Context, id string) (*logstore.MCPToolLog, error) { diff --git a/transports/bifrost-http/handlers/mcp.go b/transports/bifrost-http/handlers/mcp.go index 42446d76546..80e8b1292d1 100644 --- a/transports/bifrost-http/handlers/mcp.go +++ b/transports/bifrost-http/handlers/mcp.go @@ -7,6 +7,7 @@ import ( "encoding/json" "errors" "fmt" + "slices" "sort" "strconv" "strings" @@ -111,11 +112,86 @@ func (h *MCPHandler) getMCPClients(ctx *fasthttp.RequestCtx) { return } - limitStr := string(ctx.QueryArgs().Peek("limit")) - offsetStr := string(ctx.QueryArgs().Peek("offset")) - searchStr := string(ctx.QueryArgs().Peek("search")) + params := configstore.MCPClientsQueryParams{ + Search: string(ctx.QueryArgs().Peek("search")), + ClientID: string(ctx.QueryArgs().Peek("server")), + ConnectionTypes: parseCommaSeparated(string(ctx.QueryArgs().Peek("connection_type"))), + AuthTypes: parseCommaSeparated(string(ctx.QueryArgs().Peek("auth_type"))), + VirtualKeyIDs: parseCommaSeparated(string(ctx.QueryArgs().Peek("virtual_keys"))), + } + if b, ok, err := parseBoolQueryArg(ctx, "all_virtual_keys"); err != nil { + SendError(ctx, 400, "Invalid all_virtual_keys parameter: must be a boolean") + return + } else if ok { + params.OnlyAllVirtualKeys = b + } + // Runtime state selection (connected/disconnected) — resolved against the + // live engine inside getMCPClientsPaginated since it isn't a DB column. + states := parseCommaSeparated(string(ctx.QueryArgs().Peek("state"))) + for _, s := range states { + if s != "connected" && s != "disconnected" { + SendError(ctx, 400, "Invalid state parameter: must be 'connected' or 'disconnected'") + return + } + } - h.getMCPClientsPaginated(ctx, limitStr, offsetStr, searchStr) + if limitStr := string(ctx.QueryArgs().Peek("limit")); limitStr != "" { + n, err := strconv.Atoi(limitStr) + if err != nil { + SendError(ctx, 400, "Invalid limit parameter: must be a number") + return + } + if n < 0 { + SendError(ctx, 400, "Invalid limit parameter: must be non-negative") + return + } + params.Limit = n + } + if offsetStr := string(ctx.QueryArgs().Peek("offset")); offsetStr != "" { + n, err := strconv.Atoi(offsetStr) + if err != nil { + SendError(ctx, 400, "Invalid offset parameter: must be a number") + return + } + if n < 0 { + SendError(ctx, 400, "Invalid offset parameter: must be non-negative") + return + } + params.Offset = n + } + // Optional boolean facets — nil = no filter. Unparseable values are a hard + // error (like limit/offset) so a typo can't silently drop the filter. + if b, ok, err := parseBoolQueryArg(ctx, "code_mode"); err != nil { + SendError(ctx, 400, "Invalid code_mode parameter: must be a boolean") + return + } else if ok { + params.IsCodeModeClient = &b + } + if b, ok, err := parseBoolQueryArg(ctx, "disabled"); err != nil { + SendError(ctx, 400, "Invalid disabled parameter: must be a boolean") + return + } else if ok { + params.Disabled = &b + } + + h.getMCPClientsPaginated(ctx, params, states) +} + +// parseBoolQueryArg reads an optional boolean query parameter. It returns +// (value, true, nil) when the parameter is present and parses as a bool, +// (false, false, nil) when the parameter is absent (no filter), and +// (false, false, err) when present but unparseable — callers should surface +// the last case as an HTTP 400 rather than silently dropping the filter. +func parseBoolQueryArg(ctx *fasthttp.RequestCtx, key string) (bool, bool, error) { + raw := string(ctx.QueryArgs().Peek(key)) + if raw == "" { + return false, false, nil + } + b, err := strconv.ParseBool(raw) + if err != nil { + return false, false, err + } + return b, true, nil } // getMCPLibrary handles GET /api/mcp/library — paginated, searchable, filterable @@ -244,37 +320,43 @@ func (h *MCPHandler) forceSyncMCPLibrary(ctx *fasthttp.RequestCtx) { }) } -// getMCPClientsPaginated handles the paginated path for GET /api/mcp/clients -func (h *MCPHandler) getMCPClientsPaginated(ctx *fasthttp.RequestCtx, limitStr, offsetStr, searchStr string) { - params := configstore.MCPClientsQueryParams{ - Search: searchStr, - Limit: 100, +// getMCPClientsPaginated handles the paginated path for GET /api/mcp/clients. +// states carries the raw connection-state selection (connected/disconnected); +// it is resolved against the live engine here because state is not a DB column. +func (h *MCPHandler) getMCPClientsPaginated(ctx *fasthttp.RequestCtx, params configstore.MCPClientsQueryParams, states []string) { + // Get connected clients from Bifrost engine — used both to resolve the + // runtime state filter and to merge live state/tools onto each row below. + clientsInBifrost, err := h.client.GetMCPClients() + if err != nil { + SendError(ctx, fasthttp.StatusInternalServerError, fmt.Sprintf("Failed to get MCP clients from Bifrost: %v", err)) + return } - if limitStr != "" { - n, err := strconv.Atoi(limitStr) - if err != nil { - SendError(ctx, 400, "Invalid limit parameter: must be a number") - return - } - if n < 0 { - SendError(ctx, 400, "Invalid limit parameter: must be non-negative") - return - } - params.Limit = n + connectedClientsMap := make(map[string]schemas.MCPClient) + for _, client := range clientsInBifrost { + connectedClientsMap[client.Config.ID] = client } - if offsetStr != "" { - n, err := strconv.Atoi(offsetStr) - if err != nil { - SendError(ctx, 400, "Invalid offset parameter: must be a number") - return - } - if n < 0 { - SendError(ctx, 400, "Invalid offset parameter: must be non-negative") - return + + // Resolve the runtime state filter into a connected-id allow/block list the + // store can apply within the same paginated query. "connected" means the + // engine reports MCPConnectionStateConnected; everything else (disconnected, + // error, disabled, not-in-engine) counts as disconnected. Selecting both — + // or neither — is a no-op. + if wantConnected, wantDisconnected := slices.Contains(states, "connected"), slices.Contains(states, "disconnected"); wantConnected != wantDisconnected { + connectedIDs := make([]string, 0, len(clientsInBifrost)) + for _, c := range clientsInBifrost { + if c.State == schemas.MCPConnectionStateConnected { + connectedIDs = append(connectedIDs, c.Config.ID) + } } - params.Offset = n + params.StateClientIDs = connectedIDs + params.StateInclude = &wantConnected } + // Normalise pagination (0 → 25 default, cap 100) before the query so the + // echoed limit/offset match the rows actually returned — same helper every + // other paginated handler uses. + params.Limit, params.Offset = ClampPaginationParams(params.Limit, params.Offset) + dbClients, totalCount, err := h.store.ConfigStore.GetMCPClientsPaginated(ctx, params) if err != nil { logger.Error("failed to retrieve MCP clients: %v", err) @@ -282,17 +364,6 @@ func (h *MCPHandler) getMCPClientsPaginated(ctx *fasthttp.RequestCtx, limitStr, return } - // Get connected clients from Bifrost engine for state/tools merge - clientsInBifrost, err := h.client.GetMCPClients() - if err != nil { - SendError(ctx, fasthttp.StatusInternalServerError, fmt.Sprintf("Failed to get MCP clients from Bifrost: %v", err)) - return - } - connectedClientsMap := make(map[string]schemas.MCPClient) - for _, client := range clientsInBifrost { - connectedClientsMap[client.Config.ID] = client - } - // Batch-fetch all VK assignments for this page in a single query, then group by client ID. vkNameByID := make(map[string]string) assignmentsByClientID := make(map[uint][]configstoreTables.TableVirtualKeyMCPConfig) @@ -372,6 +443,7 @@ func (h *MCPHandler) getMCPClientsPaginated(ctx *fasthttp.RequestCtx, limitStr, AllowedExtraHeaders: dbClient.AllowedExtraHeaders, IsPingAvailable: &isPingAvailable, ToolSyncInterval: time.Duration(dbClient.ToolSyncInterval) * time.Second, + ToolExecutionTimeout: time.Duration(dbClient.ToolExecutionTimeout) * time.Second, ToolPricing: dbClient.ToolPricing, AllowOnAllVirtualKeys: dbClient.AllowOnAllVirtualKeys, Disabled: dbClient.Disabled, @@ -469,10 +541,10 @@ func (h *MCPHandler) reconnectMCPClient(ctx *fasthttp.RequestCtx) { type OAuthConfigRequest struct { ClientID *schemas.SecretVar `json:"client_id"` ClientSecret *schemas.SecretVar `json:"client_secret"` - AuthorizeURL string `json:"authorize_url"` - TokenURL string `json:"token_url"` - RegistrationURL string `json:"registration_url"` - Scopes []string `json:"scopes"` + AuthorizeURL string `json:"authorize_url"` + TokenURL string `json:"token_url"` + RegistrationURL string `json:"registration_url"` + Scopes []string `json:"scopes"` } // MCPClientRequest represents the full MCP client creation request with OAuth support. @@ -499,21 +571,22 @@ type MCPVKConfigRequest struct { // Immutable fields (connection_type, auth_type, connection_string, stdio_config) are not // accepted here; they cannot be changed after creation. type MCPClientUpdateRequest struct { - Name *string `json:"name,omitempty"` - Disabled *bool `json:"disabled,omitempty"` - AllowOnAllVirtualKeys *bool `json:"allow_on_all_virtual_keys,omitempty"` - IsCodeModeClient *bool `json:"is_code_mode_client,omitempty"` - IsPingAvailable *bool `json:"is_ping_available,omitempty"` - ToolSyncInterval *int `json:"tool_sync_interval,omitempty"` + Name *string `json:"name,omitempty"` + Disabled *bool `json:"disabled,omitempty"` + AllowOnAllVirtualKeys *bool `json:"allow_on_all_virtual_keys,omitempty"` + IsCodeModeClient *bool `json:"is_code_mode_client,omitempty"` + IsPingAvailable *bool `json:"is_ping_available,omitempty"` + ToolSyncInterval *int `json:"tool_sync_interval,omitempty"` + ToolExecutionTimeout *int `json:"tool_execution_timeout,omitempty"` Headers map[string]schemas.SecretVar `json:"headers,omitempty"` - AllowedExtraHeaders *schemas.WhiteList `json:"allowed_extra_headers,omitempty"` - ToolPricing map[string]float64 `json:"tool_pricing,omitempty"` - ToolsToExecute *schemas.WhiteList `json:"tools_to_execute,omitempty"` - ToolsToAutoExecute *schemas.WhiteList `json:"tools_to_auto_execute,omitempty"` - PerUserHeaderKeys *[]string `json:"per_user_header_keys,omitempty"` - TLSConfig *schemas.MCPTLSConfig `json:"tls_config,omitempty"` - VKConfigs *[]MCPVKConfigRequest `json:"vk_configs,omitempty"` - OauthConfig *OAuthConfigRequest `json:"oauth_config,omitempty"` + AllowedExtraHeaders *schemas.WhiteList `json:"allowed_extra_headers,omitempty"` + ToolPricing map[string]float64 `json:"tool_pricing,omitempty"` + ToolsToExecute *schemas.WhiteList `json:"tools_to_execute,omitempty"` + ToolsToAutoExecute *schemas.WhiteList `json:"tools_to_auto_execute,omitempty"` + PerUserHeaderKeys *[]string `json:"per_user_header_keys,omitempty"` + TLSConfig *schemas.MCPTLSConfig `json:"tls_config,omitempty"` + VKConfigs *[]MCPVKConfigRequest `json:"vk_configs,omitempty"` + OauthConfig *OAuthConfigRequest `json:"oauth_config,omitempty"` } // addMCPClient handles POST /api/mcp/client - Add a new MCP client @@ -527,7 +600,7 @@ func (h *MCPHandler) addMCPClient(ctx *fasthttp.RequestCtx) { var req MCPClientRequest if err := json.Unmarshal(ctx.PostBody(), &req); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid request format: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } @@ -937,7 +1010,7 @@ func (h *MCPHandler) updateMCPClient(ctx *fasthttp.RequestCtx) { } var req MCPClientUpdateRequest if err := json.Unmarshal(ctx.PostBody(), &req); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid request format: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } @@ -1024,6 +1097,14 @@ func (h *MCPHandler) updateMCPClient(ctx *fasthttp.RequestCtx) { if req.ToolSyncInterval != nil { resolvedToolSyncInterval = time.Duration(*req.ToolSyncInterval) * time.Minute } + resolvedToolExecutionTimeout := existingConfig.ToolExecutionTimeout + if req.ToolExecutionTimeout != nil { + if *req.ToolExecutionTimeout < 0 { + SendError(ctx, fasthttp.StatusBadRequest, "tool_execution_timeout must be >= 0") + return + } + resolvedToolExecutionTimeout = time.Duration(*req.ToolExecutionTimeout) * time.Second + } // Resolve tools_to_execute and tools_to_auto_execute. resolvedToolsToExecute := existingConfig.ToolsToExecute @@ -1194,6 +1275,7 @@ func (h *MCPHandler) updateMCPClient(ctx *fasthttp.RequestCtx) { IsPingAvailable: isPingAvailable, ToolPricing: toolPricing, ToolSyncInterval: int(resolvedToolSyncInterval / time.Second), + ToolExecutionTimeout: int(resolvedToolExecutionTimeout / time.Second), AuthType: string(existingConfig.AuthType), OauthConfigID: existingConfig.OauthConfigID, AllowOnAllVirtualKeys: allowOnAllVKs, @@ -1257,6 +1339,7 @@ func (h *MCPHandler) updateMCPClient(ctx *fasthttp.RequestCtx) { OauthConfigID: existingConfig.OauthConfigID, IsPingAvailable: isPingAvailable, ToolSyncInterval: toolSyncInterval, + ToolExecutionTimeout: resolvedToolExecutionTimeout, ToolPricing: toolPricing, AllowOnAllVirtualKeys: allowOnAllVKs, Disabled: disabled, @@ -1926,7 +2009,7 @@ func (h *MCPHandler) createMCPLibraryEntry(ctx *fasthttp.RequestCtx) { var req CreateMCPLibraryEntryRequest if err := json.Unmarshal(ctx.PostBody(), &req); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid request format: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } diff --git a/transports/bifrost-http/handlers/mcp_per_user_headers.go b/transports/bifrost-http/handlers/mcpheaders.go similarity index 99% rename from transports/bifrost-http/handlers/mcp_per_user_headers.go rename to transports/bifrost-http/handlers/mcpheaders.go index aba10ce9dac..d1f9eb91b37 100644 --- a/transports/bifrost-http/handlers/mcp_per_user_headers.go +++ b/transports/bifrost-http/handlers/mcpheaders.go @@ -192,7 +192,7 @@ func (h *MCPPerUserHeadersHandler) flowSubmit(ctx *fasthttp.RequestCtx) { } var req flowSubmitRequest if err := json.Unmarshal(ctx.PostBody(), &req); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid request format: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } diff --git a/transports/bifrost-http/handlers/mcpinference.go b/transports/bifrost-http/handlers/mcpinference.go index a3f8d65940e..ad4cc6dba08 100644 --- a/transports/bifrost-http/handlers/mcpinference.go +++ b/transports/bifrost-http/handlers/mcpinference.go @@ -1,7 +1,6 @@ package handlers import ( - "fmt" "strings" "github.com/bytedance/sonic" @@ -49,7 +48,7 @@ func (h *MCPInferenceHandler) executeTool(ctx *fasthttp.RequestCtx) { func (h *MCPInferenceHandler) executeChatMCPTool(ctx *fasthttp.RequestCtx) { var req schemas.ChatAssistantMessageToolCall if err := sonic.Unmarshal(ctx.PostBody(), &req); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid request format: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } @@ -82,7 +81,7 @@ func (h *MCPInferenceHandler) executeChatMCPTool(ctx *fasthttp.RequestCtx) { func (h *MCPInferenceHandler) executeResponsesMCPTool(ctx *fasthttp.RequestCtx) { var req schemas.ResponsesToolMessage if err := sonic.Unmarshal(ctx.PostBody(), &req); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid request format: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } diff --git a/transports/bifrost-http/handlers/oauth2.go b/transports/bifrost-http/handlers/mcpoauth2.go similarity index 100% rename from transports/bifrost-http/handlers/oauth2.go rename to transports/bifrost-http/handlers/mcpoauth2.go diff --git a/transports/bifrost-http/handlers/mcpoauth2consent.go b/transports/bifrost-http/handlers/mcpoauth2consent.go new file mode 100644 index 00000000000..815125a45aa --- /dev/null +++ b/transports/bifrost-http/handlers/mcpoauth2consent.go @@ -0,0 +1,432 @@ +package handlers + +import ( + "context" + "errors" + "fmt" + "net/url" + "slices" + "strings" + "time" + + "github.com/bytedance/sonic" + "github.com/fasthttp/router" + "github.com/maximhq/bifrost/core/schemas" + "github.com/maximhq/bifrost/framework/configstore" + configtables "github.com/maximhq/bifrost/framework/configstore/tables" + "github.com/maximhq/bifrost/framework/temptoken" + "github.com/maximhq/bifrost/transports/bifrost-http/lib" + "github.com/valyala/fasthttp" +) + +// OAuth2IdentityResolver is an optional extension point for user identity +// resolution during the OAuth2 consent flow. When nil, only vk and session +// modes are offered. Implementations are registered at server init time. +type OAuth2IdentityResolver interface { + // IsUserModeAvailable reports whether user mode can be offered (e.g. an + // identity provider is configured). + IsUserModeAvailable() bool + // ResolveUserIdentity reads the current session from the request context + // and returns the resolved userID and display name. The auth middleware + // populates the session into context before the handler runs (same path as + // the existing per-user upstream OAuth consent pages). Returns empty + // userID when no valid session is present. + ResolveUserIdentity(ctx *fasthttp.RequestCtx) (userID string, name string, err error) + // ResolveVKUserUpgrade checks whether a VK is bound to a specific user. + // Returns the userID if bound, empty string if unbound. + ResolveVKUserUpgrade(ctx context.Context, vkID string) (userID string, err error) + // ResolveUserVirtualKey returns the ID of a virtual key that represents the + // user's effective MCP grant, or an empty string when the user has none. + // Callers use it to scope the /mcp tool listing for user-mode tokens, which + // carry no virtual key of their own. When a user maps to several equivalent + // virtual keys, any one of them may be returned. + ResolveUserVirtualKey(ctx context.Context, userID string) (vkID string, err error) + // IsUserActive reports whether the given user identity still exists and is + // usable. Returns (false, nil) when the user is gone or deactivated; an + // error is reserved for transient lookup failures that should be retried. + // Used to cut off a deleted user's grant at request and refresh time rather + // than waiting for the access token to expire, mirroring the virtual-key + // liveness check. + IsUserActive(ctx context.Context, userID string) (active bool, err error) +} + +// OAuth2ConsentHandler serves the two consent flow APIs: +// +// - GET /api/oauth2/consent/flows/{id} — flow detail + available modes +// - PUT /api/oauth2/consent/flows/{id} — identity resolution + code mint +// +// These routes go through the standard auth middleware chain. The +// oauth2_consent temp token (embedded in the consent page URL fragment) acts +// as the credential via the middleware's temp-token fallback path — identical +// to how the existing per-user upstream OAuth consent pages work. +type OAuth2ConsentHandler struct { + store *lib.Config + tempTokens *temptoken.Service // used to invalidate the token after consent + identityResolver OAuth2IdentityResolver // optional; nil = vk + session modes only +} + +// NewOAuth2ConsentHandler creates a new consent handler. identityResolver may +// be nil — the handler degrades gracefully to vk + session modes only. +func NewOAuth2ConsentHandler(store *lib.Config, tempTokens *temptoken.Service, identityResolver OAuth2IdentityResolver) *OAuth2ConsentHandler { + return &OAuth2ConsentHandler{ + store: store, + tempTokens: tempTokens, + identityResolver: identityResolver, + } +} + +// RegisterRoutes wires the two consent routes with the provided middlewares. +func (h *OAuth2ConsentHandler) RegisterRoutes(r *router.Router, middlewares ...schemas.BifrostHTTPMiddleware) { + r.GET("/api/oauth2/consent/flows/{id}", lib.ChainMiddlewares(h.flowDetail, middlewares...)) + r.PUT("/api/oauth2/consent/flows/{id}", lib.ChainMiddlewares(h.flowSubmit, middlewares...)) +} + +// consentFlowDetailResponse is the wire shape for GET /api/oauth2/consent/flows/{id}. +type consentFlowDetailResponse struct { + ClientName string `json:"client_name"` + AvailableModes []consentFlowMode `json:"available_modes"` + LoggedInUser *loggedInUser `json:"logged_in_user,omitempty"` // non-nil when a valid session is present + ExpiresAt string `json:"expires_at"` +} + +type loggedInUser struct { + ID string `json:"id"` + Name string `json:"name,omitempty"` +} + +// GET /api/oauth2/consent/flows/{id} +func (h *OAuth2ConsentHandler) flowDetail(ctx *fasthttp.RequestCtx) { + if h.store.ConfigStore == nil { + SendError(ctx, fasthttp.StatusServiceUnavailable, "config store unavailable") + return + } + flowID, ok := ctx.UserValue("id").(string) + if !ok || flowID == "" { + SendError(ctx, fasthttp.StatusBadRequest, "invalid flow id") + return + } + req := h.loadPendingFlow(ctx, flowID) + if req == nil { + return + } + + client, err := h.store.ConfigStore.GetOAuth2ClientByClientID(ctx, req.ClientID) + if err != nil || client == nil { + if errors.Is(err, configstore.ErrNotFound) { + SendError(ctx, fasthttp.StatusInternalServerError, "client registration not found") + return + } + logger.Error("oauth2 consent: failed to load client: %v", err) + SendError(ctx, fasthttp.StatusInternalServerError, "failed to load client") + return + } + + resp := consentFlowDetailResponse{ + ClientName: client.ClientName, + AvailableModes: h.availableModes(), + ExpiresAt: req.ExpiresAt.UTC().Format(time.RFC3339), + } + + // If a valid session is present, surface the user so the consent page can + // offer "Continue as {user}" without requiring a login redirect. + if h.identityResolver != nil { + if userID, name, err := h.identityResolver.ResolveUserIdentity(ctx); err == nil && userID != "" { + resp.LoggedInUser = &loggedInUser{ID: userID, Name: name} + } + } + + SendJSON(ctx, resp) +} + +type consentFlowMode string + +const ( + consentFlowModeVK consentFlowMode = "vk" + consentFlowModeSession consentFlowMode = "session" + consentFlowModeUser consentFlowMode = "user" +) + +// consentFlowSubmitRequest is the wire shape for PUT /api/oauth2/consent/flows/{id}. +type consentFlowSubmitRequest struct { + Mode consentFlowMode `json:"mode"` // "vk" | "session" | "user" + Value string `json:"value"` // VK plaintext value for mode=vk; empty for session/user +} + +// consentFlowSubmitResponse is returned after successful consent. The frontend +// navigates to RedirectURL to complete the OAuth handshake with the MCP client. +type consentFlowSubmitResponse struct { + RedirectURL string `json:"redirect_url"` +} + +// PUT /api/oauth2/consent/flows/{id} +func (h *OAuth2ConsentHandler) flowSubmit(ctx *fasthttp.RequestCtx) { + if h.store.ConfigStore == nil { + SendError(ctx, fasthttp.StatusServiceUnavailable, "config store unavailable") + return + } + flowID, ok := ctx.UserValue("id").(string) + if !ok || flowID == "" { + SendError(ctx, fasthttp.StatusBadRequest, "invalid flow id") + return + } + req := h.loadPendingFlow(ctx, flowID) + if req == nil { + return + } + + var body consentFlowSubmitRequest + if err := sonic.Unmarshal(ctx.PostBody(), &body); err != nil { + SendError(ctx, fasthttp.StatusBadRequest, "invalid request body") + return + } + + allowed := h.availableModes() + modeAllowed := slices.Contains(allowed, body.Mode) + if !modeAllowed { + SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("mode %q is not available", body.Mode)) + return + } + + bfMode, bfSub, err := h.resolveIdentity(ctx, body) + if err != nil { + // resolveIdentity tags each failure with its HTTP status: client mistakes + // stay 400, infrastructure faults (crypto/DB) surface as 500. Fall back to + // 400 for any untagged error. + status := fasthttp.StatusBadRequest + if ce, ok := errors.AsType[*consentError](err); ok { + status = ce.status + } + SendError(ctx, status, err.Error()) + return + } + + // Mint the authorization code. Only the SHA256 hash is stored — the + // plaintext travels to the client via the redirect URI and is never + // persisted anywhere (RFC 6749 §4.1.2). + code, err := generateSecureToken(32) + if err != nil { + SendError(ctx, fasthttp.StatusInternalServerError, "failed to generate authorization code") + return + } + + req.BfMode = string(bfMode) + req.BfSub = bfSub + hash := hashSHA256Hex(code) + req.CodeHash = &hash + req.Status = configtables.OAuth2AuthorizeRequestStatusConsented + req.UpdatedAt = time.Now() + + if err := h.store.ConfigStore.ConsentOAuth2AuthorizeRequest(ctx, req); err != nil { + if errors.Is(err, configstore.ErrNotFound) { + // The flow was concurrently consented (or otherwise left pending) between + // load and write — a racing duplicate submit. Don't overwrite the code + // the winning request already minted. + SendError(ctx, fasthttp.StatusConflict, "authorization flow has already been completed") + return + } + SendError(ctx, fasthttp.StatusInternalServerError, "failed to record consent") + return + } + + // Invalidate the temp token so the consent page cannot be submitted twice. + if h.tempTokens != nil { + _, _ = h.tempTokens.DeleteByResourceID(ctx, temptoken.OAuth2ConsentScopeName, flowID) + } + + issuer := oauth2IssuerURL(ctx, h.store) + redirectURL, err := buildRedirectURL(req.RedirectURI, map[string]string{ + "code": code, + "state": req.State, + "iss": issuer, // RFC 9207: include issuer so client can validate + }) + if err != nil { + SendError(ctx, fasthttp.StatusInternalServerError, fmt.Sprintf("failed to build redirect URL: %v", err)) + return + } + + SendJSON(ctx, consentFlowSubmitResponse{RedirectURL: redirectURL}) +} + +// loadPendingFlow reads the authorize request and validates it is still pending +// and within its TTL. Writes an error response and returns nil on any failure. +func (h *OAuth2ConsentHandler) loadPendingFlow(ctx *fasthttp.RequestCtx, flowID string) *configtables.TableOAuth2AuthorizeRequest { + req, err := h.store.ConfigStore.GetOAuth2AuthorizeRequestByID(ctx, flowID) + if err != nil || req == nil { + if errors.Is(err, configstore.ErrNotFound) { + SendError(ctx, fasthttp.StatusNotFound, "authorization flow not found or expired") + return nil + } + logger.Error("oauth2 consent: failed to load flow: %v", err) + SendError(ctx, fasthttp.StatusInternalServerError, "failed to load flow") + return nil + } + if req.Status != configtables.OAuth2AuthorizeRequestStatusPending { + SendError(ctx, fasthttp.StatusGone, "authorization flow has already been completed") + return nil + } + if time.Now().After(req.ExpiresAt) { + SendError(ctx, fasthttp.StatusGone, "authorization flow has expired") + return nil + } + return req +} + +// availableModes derives which identity modes to offer based on server config. +func (h *OAuth2ConsentHandler) availableModes() []consentFlowMode { + h.store.Mu.RLock() + enforceAuth := h.store.ClientConfig.EnforceAuthOnInference + h.store.Mu.RUnlock() + // oauth2ServerCfg is nil-safe: OAuth2ServerConfig is an unset pointer until the + // AS is configured, and a raw deref here panics the request worker. It takes its + // own read lock, so it's called outside the lock above. + disableVKIdentity := oauth2ServerCfg(h.store).DisableVKIdentity + + userModeAvailable := h.identityResolver != nil && h.identityResolver.IsUserModeAvailable() + + // Virtual-key identity may be suppressed, but only when user identity is + // available — so the flow always retains the identity-provider path and can + // never be left with no usable mode. + disableVK := userModeAvailable && disableVKIdentity + + modes := []consentFlowMode{} + if !disableVK { + modes = append(modes, consentFlowModeVK) + } + if !enforceAuth { + modes = append(modes, consentFlowModeSession) + } + if userModeAvailable { + modes = append(modes, consentFlowModeUser) + } + return modes +} + +// consentError carries an HTTP status so the caller can distinguish a client +// mistake (400 — bad/inactive VK, no session) from a server-side failure +// (500 — crypto or DB error). resolveIdentity mixes both kinds; without the +// status every failure would be reported as a 400, mislabeling a broken server +// as a bad request and leaking the cause to the client. +type consentError struct { + status int + message string +} + +func (e *consentError) Error() string { return e.message } + +// clientConsentError is a 400 the caller can act on; the message is shown verbatim. +func clientConsentError(format string, args ...any) error { + return &consentError{status: fasthttp.StatusBadRequest, message: fmt.Sprintf(format, args...)} +} + +// serverConsentError is a 500 infrastructure failure. The cause is logged by the +// caller; only the generic message reaches the client. +func serverConsentError(message string) error { + return &consentError{status: fasthttp.StatusInternalServerError, message: message} +} + +// resolveIdentity resolves bf_mode and bf_sub for the submitted consent. +func (h *OAuth2ConsentHandler) resolveIdentity(ctx *fasthttp.RequestCtx, body consentFlowSubmitRequest) (schemas.MCPAuthMode, string, error) { + switch body.Mode { + case consentFlowModeVK: + return h.resolveVKIdentity(ctx, body.Value) + case consentFlowModeSession: + // Server-mints the session token — never client-asserted. + sessionToken, err := generateSecureToken(32) + if err != nil { + // crypto/rand failure is a server fault, not a bad request. + logger.Error("oauth2 consent: failed to generate session token: %v", err) + return "", "", serverConsentError("failed to generate session token") + } + return schemas.MCPAuthModeSession, sessionToken, nil + case consentFlowModeUser: + if h.identityResolver == nil { + return "", "", clientConsentError("user mode is not available") + } + userID, _, err := h.identityResolver.ResolveUserIdentity(ctx) + if err != nil { + // Keep the cause server-side — it can carry infrastructure details + // that must not leak to the client. + logger.Error("oauth2 consent: user identity resolution failed: %v", err) + return "", "", serverConsentError("failed to resolve user identity") + } + if userID == "" { + return "", "", clientConsentError("no active session; please sign in first") + } + return schemas.MCPAuthModeUser, userID, nil + default: + return "", "", clientConsentError("unknown mode %q", body.Mode) + } +} + +// resolveVKIdentity validates a VK and checks for user binding. +// If a user-bound VK is presented when an identity provider is configured, +// the logged-in user must match the VK owner before the upgrade proceeds. +func (h *OAuth2ConsentHandler) resolveVKIdentity(ctx *fasthttp.RequestCtx, vkValue string) (schemas.MCPAuthMode, string, error) { + if strings.TrimSpace(vkValue) == "" { + return "", "", clientConsentError("virtual key value is required") + } + + vk, err := h.store.ConfigStore.GetVirtualKeyByValue(ctx, vkValue) + if err != nil { + if errors.Is(err, configstore.ErrNotFound) { + return "", "", clientConsentError("virtual key not found") + } + // A non-NotFound lookup error is an infrastructure fault - log the cause + // and return a generic 500. + logger.Error("oauth2 consent: failed to load virtual key: %v", err) + return "", "", serverConsentError("failed to validate virtual key") + } + if !vk.IsActiveValue() { + return "", "", clientConsentError("virtual key is inactive") + } + + // Check for a VK→user binding. If the VK is bound to a specific user, + // upgrade the identity so the JWT carries the user ID — this ensures + // per-user upstream OAuth tokens are unified under a single identity + // regardless of whether auth happened via VK or direct user login. + if h.identityResolver != nil { + userID, err := h.identityResolver.ResolveVKUserUpgrade(ctx, vk.ID) + if err != nil { + // Keep the cause server-side — a DB/driver error can carry table or + // connection details that must not leak to the client. + logger.Error("oauth2 consent: VK user binding lookup failed: %v", err) + return "", "", serverConsentError("failed to check VK user binding") + } + if userID != "" { + // VK is user-bound. Verify the currently logged-in user matches the + // VK owner before upgrading — possession of a user-bound VK alone is + // not sufficient proof of identity when session auth is available. + if h.identityResolver.IsUserModeAvailable() { + loggedInUserID, _, sessionErr := h.identityResolver.ResolveUserIdentity(ctx) + if sessionErr != nil || loggedInUserID == "" { + return "", "", clientConsentError("this key belongs to a specific user; please sign in to continue") + } + if loggedInUserID != userID { + return "", "", clientConsentError("this key belongs to a different user than the currently signed-in user") + } + } + return schemas.MCPAuthModeUser, userID, nil + } + } + + return schemas.MCPAuthModeVK, vk.ID, nil +} + +// buildRedirectURL appends query params to a base URL using net/url so +// existing query params are preserved and values are correctly percent-encoded. +// Returns an error when base cannot be parsed so the caller can return a 500 +// before the temp token and authorization code become irrecoverable. +func buildRedirectURL(base string, params map[string]string) (string, error) { + u, err := url.Parse(base) + if err != nil { + return "", fmt.Errorf("invalid redirect URI: %w", err) + } + q := u.Query() + for k, v := range params { + if v != "" { + q.Set(k, v) + } + } + u.RawQuery = q.Encode() + return u.String(), nil +} diff --git a/transports/bifrost-http/handlers/mcpoauth2consent_test.go b/transports/bifrost-http/handlers/mcpoauth2consent_test.go new file mode 100644 index 00000000000..04ba102f5c4 --- /dev/null +++ b/transports/bifrost-http/handlers/mcpoauth2consent_test.go @@ -0,0 +1,309 @@ +package handlers + +import ( + "context" + "encoding/json" + "testing" + "time" + + "github.com/maximhq/bifrost/core/schemas" + configtables "github.com/maximhq/bifrost/framework/configstore/tables" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "github.com/valyala/fasthttp" +) + +// fakeResolver is a configurable OAuth2IdentityResolver for consent tests. +type fakeResolver struct { + userModeAvailable bool + userID string + name string + resolveErr error + vkBoundUserID string + vkBindErr error + userVKID string + userVKErr error + userInactive bool // when true, IsUserActive reports the user as gone + userActiveErr error +} + +func (f *fakeResolver) IsUserModeAvailable() bool { return f.userModeAvailable } +func (f *fakeResolver) ResolveUserIdentity(_ *fasthttp.RequestCtx) (string, string, error) { + return f.userID, f.name, f.resolveErr +} +func (f *fakeResolver) ResolveVKUserUpgrade(_ context.Context, _ string) (string, error) { + return f.vkBoundUserID, f.vkBindErr +} +func (f *fakeResolver) ResolveUserVirtualKey(_ context.Context, _ string) (string, error) { + return f.userVKID, f.userVKErr +} +func (f *fakeResolver) IsUserActive(_ context.Context, _ string) (bool, error) { + return !f.userInactive, f.userActiveErr +} + +func newConsentStore() *mockOAuth2Store { + return &mockOAuth2Store{ + vksByValue: map[string]*configtables.TableVirtualKey{}, + clients: map[string]*configtables.TableOAuth2Client{ + "client-1": {ClientID: "client-1", ClientName: "Test Client"}, + }, + authReqs: map[string]*configtables.TableOAuth2AuthorizeRequest{}, + } +} + +func seedPendingFlow(store *mockOAuth2Store, id string, expires time.Time) { + store.authReqs[id] = &configtables.TableOAuth2AuthorizeRequest{ + ID: id, + ClientID: "client-1", + RedirectURI: "http://127.0.0.1/cb", + State: "st", + Status: configtables.OAuth2AuthorizeRequestStatusPending, + ExpiresAt: expires, + } +} + +func newConsentHandler(store *mockOAuth2Store, resolver OAuth2IdentityResolver, enforceAuth bool) *OAuth2ConsentHandler { + SetLogger(&mockLogger{}) + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, enforceAuth) + return NewOAuth2ConsentHandler(cfg, nil, resolver) +} + +func consentCtx(flowID, body string) *fasthttp.RequestCtx { + ctx := &fasthttp.RequestCtx{} + ctx.SetUserValue("id", flowID) + if body != "" { + ctx.Request.SetBody([]byte(body)) + } + return ctx +} + +func TestConsentFlowDetail(t *testing.T) { + t.Run("pending flow returns client and modes", func(t *testing.T) { + store := newConsentStore() + seedPendingFlow(store, "flow-1", time.Now().Add(time.Minute)) + h := newConsentHandler(store, nil, false) + ctx := consentCtx("flow-1", "") + h.flowDetail(ctx) + require.Equal(t, fasthttp.StatusOK, ctx.Response.StatusCode()) + + var resp consentFlowDetailResponse + require.NoError(t, json.Unmarshal(ctx.Response.Body(), &resp)) + assert.Equal(t, "Test Client", resp.ClientName) + assert.Equal(t, []consentFlowMode{consentFlowModeVK, consentFlowModeSession}, resp.AvailableModes) + }) + + t.Run("missing flow returns 404", func(t *testing.T) { + h := newConsentHandler(newConsentStore(), nil, false) + ctx := consentCtx("nope", "") + h.flowDetail(ctx) + assert.Equal(t, fasthttp.StatusNotFound, ctx.Response.StatusCode()) + }) + + t.Run("empty flow id returns 400", func(t *testing.T) { + h := newConsentHandler(newConsentStore(), nil, false) + ctx := consentCtx("", "") + h.flowDetail(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + }) + + t.Run("expired flow returns 410", func(t *testing.T) { + store := newConsentStore() + seedPendingFlow(store, "flow-1", time.Now().Add(-time.Minute)) + h := newConsentHandler(store, nil, false) + ctx := consentCtx("flow-1", "") + h.flowDetail(ctx) + assert.Equal(t, fasthttp.StatusGone, ctx.Response.StatusCode()) + }) + + t.Run("already-consented flow returns 410", func(t *testing.T) { + store := newConsentStore() + seedPendingFlow(store, "flow-1", time.Now().Add(time.Minute)) + store.authReqs["flow-1"].Status = configtables.OAuth2AuthorizeRequestStatusConsented + h := newConsentHandler(store, nil, false) + ctx := consentCtx("flow-1", "") + h.flowDetail(ctx) + assert.Equal(t, fasthttp.StatusGone, ctx.Response.StatusCode()) + }) +} + +func TestConsentAvailableModes(t *testing.T) { + cases := []struct { + name string + resolver OAuth2IdentityResolver + enforceAuth bool + disableVK bool + want []consentFlowMode + }{ + {"vk and session when auth not enforced", nil, false, false, []consentFlowMode{consentFlowModeVK, consentFlowModeSession}}, + {"vk only when auth enforced", nil, true, false, []consentFlowMode{consentFlowModeVK}}, + {"adds user when resolver offers it", &fakeResolver{userModeAvailable: true}, false, false, []consentFlowMode{consentFlowModeVK, consentFlowModeSession, consentFlowModeUser}}, + // DisableVKIdentity drops vk, but only when user mode is available so the + // flow always keeps a usable identity path. + {"disable vk drops vk when user mode available", &fakeResolver{userModeAvailable: true}, false, true, []consentFlowMode{consentFlowModeSession, consentFlowModeUser}}, + {"disable vk leaves user-only when auth enforced", &fakeResolver{userModeAvailable: true}, true, true, []consentFlowMode{consentFlowModeUser}}, + {"disable vk ignored without user mode", nil, false, true, []consentFlowMode{consentFlowModeVK, consentFlowModeSession}}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + h := newConsentHandler(newConsentStore(), tc.resolver, tc.enforceAuth) + h.store.ClientConfig.OAuth2ServerConfig.DisableVKIdentity = tc.disableVK + assert.Equal(t, tc.want, h.availableModes()) + }) + } +} + +func TestConsentFlowSubmit_VK(t *testing.T) { + activeVK := &configtables.TableVirtualKey{ID: "vk-row-1", Value: *schemas.NewSecretVar("sk-bf-active"), IsActive: new(true)} + inactiveVK := &configtables.TableVirtualKey{ID: "vk-row-2", Value: *schemas.NewSecretVar("sk-bf-inactive"), IsActive: new(false)} + + t.Run("active VK mints a code", func(t *testing.T) { + store := newConsentStore() + store.vksByValue["sk-bf-active"] = activeVK + seedPendingFlow(store, "flow-1", time.Now().Add(time.Minute)) + h := newConsentHandler(store, nil, false) + ctx := consentCtx("flow-1", `{"mode":"vk","value":"sk-bf-active"}`) + h.flowSubmit(ctx) + require.Equal(t, fasthttp.StatusOK, ctx.Response.StatusCode(), string(ctx.Response.Body())) + + var resp consentFlowSubmitResponse + require.NoError(t, json.Unmarshal(ctx.Response.Body(), &resp)) + assert.Contains(t, resp.RedirectURL, "code=") + assert.Contains(t, resp.RedirectURL, "state=st") + // The flow is now consented and bound to the VK row id. + assert.Equal(t, "vk", store.authReqs["flow-1"].BfMode) + assert.Equal(t, "vk-row-1", store.authReqs["flow-1"].BfSub) + }) + + t.Run("inactive VK is rejected", func(t *testing.T) { + store := newConsentStore() + store.vksByValue[inactiveVK.Value.GetValue()] = inactiveVK + seedPendingFlow(store, "flow-1", time.Now().Add(time.Minute)) + h := newConsentHandler(store, nil, false) + ctx := consentCtx("flow-1", `{"mode":"vk","value":"sk-bf-inactive"}`) + h.flowSubmit(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + }) + + t.Run("unknown VK is rejected", func(t *testing.T) { + store := newConsentStore() + seedPendingFlow(store, "flow-1", time.Now().Add(time.Minute)) + h := newConsentHandler(store, nil, false) + ctx := consentCtx("flow-1", `{"mode":"vk","value":"sk-bf-missing"}`) + h.flowSubmit(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + }) + + t.Run("empty VK value is rejected", func(t *testing.T) { + store := newConsentStore() + seedPendingFlow(store, "flow-1", time.Now().Add(time.Minute)) + h := newConsentHandler(store, nil, false) + ctx := consentCtx("flow-1", `{"mode":"vk","value":""}`) + h.flowSubmit(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + }) + + t.Run("double submit returns 410 on the second attempt", func(t *testing.T) { + store := newConsentStore() + store.vksByValue[activeVK.Value.GetValue()] = activeVK + seedPendingFlow(store, "flow-1", time.Now().Add(time.Minute)) + h := newConsentHandler(store, nil, false) + + first := consentCtx("flow-1", `{"mode":"vk","value":"sk-bf-active"}`) + h.flowSubmit(first) + require.Equal(t, fasthttp.StatusOK, first.Response.StatusCode()) + + second := consentCtx("flow-1", `{"mode":"vk","value":"sk-bf-active"}`) + h.flowSubmit(second) + assert.Equal(t, fasthttp.StatusGone, second.Response.StatusCode()) + }) +} + +func TestConsentFlowSubmit_Session(t *testing.T) { + t.Run("session mode mints a server-side token when auth not enforced", func(t *testing.T) { + store := newConsentStore() + seedPendingFlow(store, "flow-1", time.Now().Add(time.Minute)) + h := newConsentHandler(store, nil, false) + ctx := consentCtx("flow-1", `{"mode":"session"}`) + h.flowSubmit(ctx) + require.Equal(t, fasthttp.StatusOK, ctx.Response.StatusCode(), string(ctx.Response.Body())) + assert.Equal(t, "session", store.authReqs["flow-1"].BfMode) + assert.NotEmpty(t, store.authReqs["flow-1"].BfSub) // server-minted, not client-asserted + }) + + t.Run("session mode is unavailable when auth is enforced", func(t *testing.T) { + store := newConsentStore() + seedPendingFlow(store, "flow-1", time.Now().Add(time.Minute)) + h := newConsentHandler(store, nil, true) + ctx := consentCtx("flow-1", `{"mode":"session"}`) + h.flowSubmit(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "not available") + }) +} + +func TestConsentFlowSubmit_User(t *testing.T) { + t.Run("user mode rejected when no resolver (mode not offered)", func(t *testing.T) { + store := newConsentStore() + seedPendingFlow(store, "flow-1", time.Now().Add(time.Minute)) + h := newConsentHandler(store, nil, false) + ctx := consentCtx("flow-1", `{"mode":"user"}`) + h.flowSubmit(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + }) + + t.Run("resolved session yields user mode", func(t *testing.T) { + store := newConsentStore() + seedPendingFlow(store, "flow-1", time.Now().Add(time.Minute)) + h := newConsentHandler(store, &fakeResolver{userModeAvailable: true, userID: "user-1", name: "Alice"}, false) + ctx := consentCtx("flow-1", `{"mode":"user"}`) + h.flowSubmit(ctx) + require.Equal(t, fasthttp.StatusOK, ctx.Response.StatusCode(), string(ctx.Response.Body())) + assert.Equal(t, "user", store.authReqs["flow-1"].BfMode) + assert.Equal(t, "user-1", store.authReqs["flow-1"].BfSub) + }) + + t.Run("user mode without a session is rejected", func(t *testing.T) { + store := newConsentStore() + seedPendingFlow(store, "flow-1", time.Now().Add(time.Minute)) + h := newConsentHandler(store, &fakeResolver{userModeAvailable: true, userID: ""}, false) + ctx := consentCtx("flow-1", `{"mode":"user"}`) + h.flowSubmit(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + }) +} + +func TestConsentFlowSubmit_VKUserBinding(t *testing.T) { + boundVK := &configtables.TableVirtualKey{ID: "vk-row-1", Value: *schemas.NewSecretVar("sk-bf-bound"), IsActive: new(true)} + + t.Run("bound VK upgrades to user when logged-in user matches", func(t *testing.T) { + store := newConsentStore() + store.vksByValue[boundVK.Value.GetValue()] = boundVK + seedPendingFlow(store, "flow-1", time.Now().Add(time.Minute)) + h := newConsentHandler(store, &fakeResolver{userModeAvailable: true, userID: "owner-1", vkBoundUserID: "owner-1"}, false) + ctx := consentCtx("flow-1", `{"mode":"vk","value":"sk-bf-bound"}`) + h.flowSubmit(ctx) + require.Equal(t, fasthttp.StatusOK, ctx.Response.StatusCode(), string(ctx.Response.Body())) + assert.Equal(t, "user", store.authReqs["flow-1"].BfMode) + assert.Equal(t, "owner-1", store.authReqs["flow-1"].BfSub) + }) + + t.Run("bound VK rejected when logged-in user differs", func(t *testing.T) { + store := newConsentStore() + store.vksByValue[boundVK.Value.GetValue()] = boundVK + seedPendingFlow(store, "flow-1", time.Now().Add(time.Minute)) + h := newConsentHandler(store, &fakeResolver{userModeAvailable: true, userID: "intruder", vkBoundUserID: "owner-1"}, false) + ctx := consentCtx("flow-1", `{"mode":"vk","value":"sk-bf-bound"}`) + h.flowSubmit(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + }) + + t.Run("bound VK rejected when not signed in", func(t *testing.T) { + store := newConsentStore() + store.vksByValue[boundVK.Value.GetValue()] = boundVK + seedPendingFlow(store, "flow-1", time.Now().Add(time.Minute)) + h := newConsentHandler(store, &fakeResolver{userModeAvailable: true, userID: "", vkBoundUserID: "owner-1"}, false) + ctx := consentCtx("flow-1", `{"mode":"vk","value":"sk-bf-bound"}`) + h.flowSubmit(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + }) +} diff --git a/transports/bifrost-http/handlers/mcpoauth2discovery.go b/transports/bifrost-http/handlers/mcpoauth2discovery.go new file mode 100644 index 00000000000..fad1f7bff83 --- /dev/null +++ b/transports/bifrost-http/handlers/mcpoauth2discovery.go @@ -0,0 +1,183 @@ +package handlers + +import ( + "crypto/rsa" + "crypto/x509" + "encoding/base64" + "encoding/pem" + "fmt" + "math/big" + + "github.com/bytedance/sonic" + "github.com/fasthttp/router" + "github.com/maximhq/bifrost/core/schemas" + "github.com/maximhq/bifrost/transports/bifrost-http/lib" + "github.com/valyala/fasthttp" +) + +// OAuth2DiscoveryHandler serves the three well-known discovery endpoints that +// make Bifrost's /mcp endpoint a spec-compliant OAuth 2.1 protected resource: +// +// - GET /.well-known/oauth-protected-resource[/{path}] (RFC 9728 PRM) +// - GET /.well-known/oauth-authorization-server[/{path}] (RFC 8414 AS metadata) +// - GET /.well-known/jwks.json (RFC 7517 JWKS) +// +// All three return 404 when MCPServerAuthMode == "headers" (the default), so +// discovery is only available when the operator explicitly enables OAuth mode. +// Discoverability is the feature toggle. +type OAuth2DiscoveryHandler struct { + store *lib.Config +} + +// NewOAuth2DiscoveryHandler creates a new discovery handler. +func NewOAuth2DiscoveryHandler(store *lib.Config) *OAuth2DiscoveryHandler { + return &OAuth2DiscoveryHandler{store: store} +} + +// RegisterRoutes wires all well-known discovery routes. Routes are always +// registered; individual handlers 404 when discovery is disabled. +func (h *OAuth2DiscoveryHandler) RegisterRoutes(r *router.Router, middlewares ...schemas.BifrostHTTPMiddleware) { + // RFC 9728: both root and path-aware well-known forms are required. + r.GET("/.well-known/oauth-protected-resource", lib.ChainMiddlewares(h.handlePRM, middlewares...)) + r.GET("/.well-known/oauth-protected-resource/{path:*}", lib.ChainMiddlewares(h.handlePRM, middlewares...)) + + // RFC 8414: same two forms. + r.GET("/.well-known/oauth-authorization-server", lib.ChainMiddlewares(h.handleASMetadata, middlewares...)) + r.GET("/.well-known/oauth-authorization-server/{path:*}", lib.ChainMiddlewares(h.handleASMetadata, middlewares...)) + + // RFC 7517 JWKS. + r.GET("/.well-known/jwks.json", lib.ChainMiddlewares(h.handleJWKS, middlewares...)) +} + +// discoveryEnabled reports whether OAuth discovery is active, reading the mode +// from the in-memory ClientConfig under the read lock. +func (h *OAuth2DiscoveryHandler) discoveryEnabled() bool { + h.store.Mu.RLock() + enabled := h.store.ClientConfig.IsMCPOAuthDiscoveryEnabled() + h.store.Mu.RUnlock() + return enabled +} + +// handlePRM serves GET /.well-known/oauth-protected-resource[/{path}]. +// RFC 9728 §3 Protected Resource Metadata. +func (h *OAuth2DiscoveryHandler) handlePRM(ctx *fasthttp.RequestCtx) { + if !h.discoveryEnabled() { + ctx.SetStatusCode(fasthttp.StatusNotFound) + return + } + + base := oauth2IssuerURL(ctx, h.store) + doc := map[string]any{ + "resource": oauth2MCPResourceURL(ctx, h.store), + "authorization_servers": []string{base}, + "scopes_supported": []string{"mcp"}, + "bearer_methods_supported": []string{"header"}, + } + data, err := sonic.Marshal(doc) + if err != nil { + SendError(ctx, fasthttp.StatusInternalServerError, fmt.Sprintf("marshal protected resource metadata: %v", err)) + return + } + ctx.SetContentType("application/json") + ctx.SetBody(data) +} + +// handleASMetadata serves GET /.well-known/oauth-authorization-server[/{path}]. +// RFC 8414 Authorization Server Metadata. +func (h *OAuth2DiscoveryHandler) handleASMetadata(ctx *fasthttp.RequestCtx) { + if !h.discoveryEnabled() { + ctx.SetStatusCode(fasthttp.StatusNotFound) + return + } + + base := oauth2IssuerURL(ctx, h.store) + doc := map[string]any{ + "issuer": base, + "authorization_endpoint": base + "/oauth2/authorize", + "token_endpoint": base + "/oauth2/token", + "registration_endpoint": base + "/oauth2/register", + "jwks_uri": base + "/.well-known/jwks.json", + "response_types_supported": []string{"code"}, + "grant_types_supported": []string{"authorization_code", "refresh_token"}, + "code_challenge_methods_supported": []string{"S256"}, + "token_endpoint_auth_methods_supported": []string{"none"}, + "scopes_supported": []string{"mcp"}, + // RFC 9207: we include iss in authorization responses. + "authorization_response_iss_parameter_supported": true, + } + data, err := sonic.Marshal(doc) + if err != nil { + SendError(ctx, fasthttp.StatusInternalServerError, fmt.Sprintf("marshal authorization server metadata: %v", err)) + return + } + ctx.SetContentType("application/json") + ctx.SetBody(data) +} + +// handleJWKS serves GET /.well-known/jwks.json (RFC 7517). +func (h *OAuth2DiscoveryHandler) handleJWKS(ctx *fasthttp.RequestCtx) { + if !h.discoveryEnabled() { + ctx.SetStatusCode(fasthttp.StatusNotFound) + return + } + + if h.store.ConfigStore == nil { + ctx.SetContentType("application/json") + ctx.SetBodyString(`{"keys":[]}`) + return + } + + key, err := h.store.GetOAuth2SigningKey(ctx) + if err != nil { + logger.Error("oauth2 discovery: failed to load signing key: %v", err) + SendError(ctx, fasthttp.StatusInternalServerError, "failed to load signing key") + return + } + + var jwks []map[string]any + if key != nil { + pub, parseErr := parseRSAPublicKeyPEM(key.PublicKeyPEM) + if parseErr != nil { + SendError(ctx, fasthttp.StatusInternalServerError, fmt.Sprintf("failed to parse signing key: %v", parseErr)) + return + } + jwks = []map[string]any{rsaPublicKeyToJWK(key.KID, "RS256", pub)} + } + + data, err := sonic.Marshal(map[string]any{"keys": jwks}) + if err != nil { + SendError(ctx, fasthttp.StatusInternalServerError, fmt.Sprintf("marshal jwks: %v", err)) + return + } + ctx.SetContentType("application/json") + ctx.SetBody(data) +} + +// parseRSAPublicKeyPEM decodes a PEM-encoded RSA public key. +func parseRSAPublicKeyPEM(pemStr string) (*rsa.PublicKey, error) { + block, rest := pem.Decode([]byte(pemStr)) + if block == nil || len(rest) > 0 { + return nil, fmt.Errorf("malformed public key PEM") + } + pub, err := x509.ParsePKIXPublicKey(block.Bytes) + if err != nil { + return nil, fmt.Errorf("parse public key: %w", err) + } + rsaPub, ok := pub.(*rsa.PublicKey) + if !ok { + return nil, fmt.Errorf("expected RSA public key, got %T", pub) + } + return rsaPub, nil +} + +// rsaPublicKeyToJWK encodes an RSA public key as a JWK (RFC 7517 §6.3). +func rsaPublicKeyToJWK(kid, alg string, pub *rsa.PublicKey) map[string]any { + return map[string]any{ + "kty": "RSA", + "use": "sig", + "kid": kid, + "alg": alg, + "n": base64.RawURLEncoding.EncodeToString(pub.N.Bytes()), + "e": base64.RawURLEncoding.EncodeToString(big.NewInt(int64(pub.E)).Bytes()), + } +} diff --git a/transports/bifrost-http/handlers/mcpoauth2discovery_test.go b/transports/bifrost-http/handlers/mcpoauth2discovery_test.go new file mode 100644 index 00000000000..edab5aad286 --- /dev/null +++ b/transports/bifrost-http/handlers/mcpoauth2discovery_test.go @@ -0,0 +1,91 @@ +package handlers + +import ( + "encoding/json" + "testing" + + configtables "github.com/maximhq/bifrost/framework/configstore/tables" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "github.com/valyala/fasthttp" +) + +func TestDiscovery_GatedOnAuthMode(t *testing.T) { + key, _ := newTestSigningKey(t) + store := &mockOAuth2Store{signingKey: key} + + t.Run("headers mode returns 404 on all discovery endpoints", func(t *testing.T) { + h := NewOAuth2DiscoveryHandler(newTestOAuth2Config(store, configtables.MCPServerAuthModeHeaders, false)) + for _, fn := range []func(*fasthttp.RequestCtx){h.handlePRM, h.handleASMetadata, h.handleJWKS} { + ctx := &fasthttp.RequestCtx{} + fn(ctx) + assert.Equal(t, fasthttp.StatusNotFound, ctx.Response.StatusCode()) + } + }) + + t.Run("oauth and both modes serve discovery", func(t *testing.T) { + for _, mode := range []configtables.MCPServerAuthMode{configtables.MCPServerAuthModeBoth, configtables.MCPServerAuthModeOAuth} { + h := NewOAuth2DiscoveryHandler(newTestOAuth2Config(store, mode, false)) + for _, fn := range []func(*fasthttp.RequestCtx){h.handlePRM, h.handleASMetadata, h.handleJWKS} { + ctx := &fasthttp.RequestCtx{} + fn(ctx) + assert.Equal(t, fasthttp.StatusOK, ctx.Response.StatusCode(), string(mode)) + } + } + }) +} + +func TestDiscovery_ProtectedResourceMetadata(t *testing.T) { + store := &mockOAuth2Store{} + h := NewOAuth2DiscoveryHandler(newTestOAuth2Config(store, configtables.MCPServerAuthModeOAuth, false)) + ctx := &fasthttp.RequestCtx{} + h.handlePRM(ctx) + require.Equal(t, fasthttp.StatusOK, ctx.Response.StatusCode()) + + var doc map[string]any + require.NoError(t, json.Unmarshal(ctx.Response.Body(), &doc)) + assert.Equal(t, testMCPResource, doc["resource"]) + assert.Equal(t, []any{testIssuer}, doc["authorization_servers"]) +} + +func TestDiscovery_AuthorizationServerMetadata(t *testing.T) { + store := &mockOAuth2Store{} + h := NewOAuth2DiscoveryHandler(newTestOAuth2Config(store, configtables.MCPServerAuthModeOAuth, false)) + ctx := &fasthttp.RequestCtx{} + h.handleASMetadata(ctx) + require.Equal(t, fasthttp.StatusOK, ctx.Response.StatusCode()) + + var doc map[string]any + require.NoError(t, json.Unmarshal(ctx.Response.Body(), &doc)) + assert.Equal(t, testIssuer, doc["issuer"]) + assert.Equal(t, testIssuer+"/oauth2/authorize", doc["authorization_endpoint"]) + assert.Equal(t, testIssuer+"/oauth2/token", doc["token_endpoint"]) + assert.Equal(t, testIssuer+"/oauth2/register", doc["registration_endpoint"]) + assert.Equal(t, []any{"code"}, doc["response_types_supported"]) + assert.Equal(t, []any{"authorization_code", "refresh_token"}, doc["grant_types_supported"]) + assert.Equal(t, []any{"S256"}, doc["code_challenge_methods_supported"]) + assert.Equal(t, []any{"none"}, doc["token_endpoint_auth_methods_supported"]) + assert.Equal(t, true, doc["authorization_response_iss_parameter_supported"]) +} + +func TestDiscovery_JWKS(t *testing.T) { + key, _ := newTestSigningKey(t) + store := &mockOAuth2Store{signingKey: key} + h := NewOAuth2DiscoveryHandler(newTestOAuth2Config(store, configtables.MCPServerAuthModeOAuth, false)) + ctx := &fasthttp.RequestCtx{} + h.handleJWKS(ctx) + require.Equal(t, fasthttp.StatusOK, ctx.Response.StatusCode()) + + var doc struct { + Keys []map[string]any `json:"keys"` + } + require.NoError(t, json.Unmarshal(ctx.Response.Body(), &doc)) + require.Len(t, doc.Keys, 1) + jwk := doc.Keys[0] + assert.Equal(t, "RSA", jwk["kty"]) + assert.Equal(t, "RS256", jwk["alg"]) + assert.Equal(t, "sig", jwk["use"]) + assert.Equal(t, key.KID, jwk["kid"]) + assert.NotEmpty(t, jwk["n"]) + assert.NotEmpty(t, jwk["e"]) +} diff --git a/transports/bifrost-http/handlers/mcpoauth2issuance.go b/transports/bifrost-http/handlers/mcpoauth2issuance.go new file mode 100644 index 00000000000..78ed4fed51c --- /dev/null +++ b/transports/bifrost-http/handlers/mcpoauth2issuance.go @@ -0,0 +1,789 @@ +package handlers + +import ( + "crypto/rand" + "crypto/rsa" + "crypto/sha256" + "crypto/subtle" + "crypto/x509" + "encoding/base64" + "encoding/hex" + "encoding/pem" + "errors" + "fmt" + "net" + "net/url" + "strings" + "time" + + "github.com/bytedance/sonic" + "github.com/fasthttp/router" + "github.com/golang-jwt/jwt/v5" + "github.com/google/uuid" + "github.com/maximhq/bifrost/core/schemas" + "github.com/maximhq/bifrost/framework/configstore" + configtables "github.com/maximhq/bifrost/framework/configstore/tables" + "github.com/maximhq/bifrost/framework/temptoken" + "github.com/maximhq/bifrost/transports/bifrost-http/lib" + "github.com/valyala/fasthttp" +) + +// OAuth2IssuanceHandler implements the three downstream OAuth2 endpoints: +// +// - POST /oauth2/register — RFC 7591 Dynamic Client Registration +// - GET /oauth2/authorize — Authorization endpoint (PKCE-S256, RFC 8707) +// - POST /oauth2/token — Token endpoint (auth-code + refresh grants) +type OAuth2IssuanceHandler struct { + store *lib.Config + tempTokens *temptoken.Service // optional; nil = no consent temp-token minted + // identityResolver mirrors the consent handler's resolver. It is consulted + // only to decide whether DisableVKIdentity is in effect (it is honored only + // when user identity is available). May be nil — VK refresh is never blocked + // then, matching availableModes which keeps offering vk without a resolver. + identityResolver OAuth2IdentityResolver +} + +// NewOAuth2IssuanceHandler creates a new issuance handler. identityResolver may +// be nil — the VK-refresh cutoff degrades to a no-op, consistent with the +// consent handler offering vk when no resolver is present. +func NewOAuth2IssuanceHandler(store *lib.Config, tempTokens *temptoken.Service, identityResolver OAuth2IdentityResolver) *OAuth2IssuanceHandler { + return &OAuth2IssuanceHandler{store: store, tempTokens: tempTokens, identityResolver: identityResolver} +} + +// RegisterRoutes wires the three OAuth2 issuance routes. +func (h *OAuth2IssuanceHandler) RegisterRoutes(r *router.Router, middlewares ...schemas.BifrostHTTPMiddleware) { + // These routes are public — no auth middleware applied. + r.POST("/oauth2/register", h.handleRegister) + r.GET("/oauth2/authorize", h.handleAuthorize) + r.POST("/oauth2/token", h.handleToken) +} + +// --- POST /oauth2/register (RFC 7591 DCR) --- + +type dcrRequest struct { + ClientName string `json:"client_name"` + RedirectURIs []string `json:"redirect_uris"` + GrantTypes []string `json:"grant_types"` + ResponseTypes []string `json:"response_types"` + TokenEndpointAuthMethod string `json:"token_endpoint_auth_method"` + Scope string `json:"scope"` +} + +func (h *OAuth2IssuanceHandler) handleRegister(ctx *fasthttp.RequestCtx) { + if h.store.ConfigStore == nil { + sendOAuthError(ctx, fasthttp.StatusServiceUnavailable, "server_error", "config store unavailable") + return + } + + var req dcrRequest + if err := sonic.Unmarshal(ctx.PostBody(), &req); err != nil { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_request", "malformed request body") + return + } + if len(req.RedirectURIs) == 0 { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_redirect_uri", "redirect_uris is required") + return + } + // Registration is public and unauthenticated, so reject dangerous schemes + // (javascript:, data:, etc.) here at the source. Only https is allowed, with + // http permitted exclusively for loopback addresses (RFC 9700 §4.1.3). + for _, uri := range req.RedirectURIs { + if !isAllowedRedirectScheme(uri) { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_redirect_uri", "redirect_uris must use https (or http for loopback addresses)") + return + } + } + // Only public clients supported. + if req.TokenEndpointAuthMethod != "" && req.TokenEndpointAuthMethod != "none" { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_client_metadata", "only token_endpoint_auth_method=none is supported") + return + } + + grantTypes := req.GrantTypes + if len(grantTypes) == 0 { + grantTypes = []string{"authorization_code"} + } + // The token endpoint only implements the authorization-code and refresh-token + // grants. Reject anything else here so a registration never advertises a flow + // that would later be refused at /oauth2/token. + for _, gt := range grantTypes { + if gt != "authorization_code" && gt != "refresh_token" { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_client_metadata", "unsupported grant_type") + return + } + } + responseTypes := req.ResponseTypes + if len(responseTypes) == 0 { + responseTypes = []string{"code"} + } + // The authorize endpoint only implements response_type=code. + for _, rt := range responseTypes { + if rt != "code" { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_client_metadata", "unsupported response_type") + return + } + } + scope := req.Scope + if scope == "" { + scope = "mcp" + } + + clientID := uuid.New().String() + client := &configtables.TableOAuth2Client{ + ID: uuid.New().String(), + ClientID: clientID, + ClientName: req.ClientName, + RedirectURIs: req.RedirectURIs, + GrantTypes: grantTypes, + Scope: scope, + CreatedAt: time.Now(), + } + if err := h.store.ConfigStore.CreateOAuth2Client(ctx, client); err != nil { + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "failed to register client") + return + } + + ctx.SetStatusCode(fasthttp.StatusCreated) + ctx.SetContentType("application/json") + data, err := sonic.Marshal(map[string]any{ + "client_id": clientID, + "client_id_issued_at": client.CreatedAt.Unix(), + "grant_types": grantTypes, + "response_types": responseTypes, + "redirect_uris": req.RedirectURIs, + "token_endpoint_auth_method": "none", + "scope": scope, + }) + if err != nil { + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "failed to marshal response") + return + } + ctx.SetBody(data) +} + +// --- GET /oauth2/authorize --- + +func (h *OAuth2IssuanceHandler) handleAuthorize(ctx *fasthttp.RequestCtx) { + if h.store.ConfigStore == nil { + sendOAuthError(ctx, fasthttp.StatusServiceUnavailable, "server_error", "config store unavailable") + return + } + + q := ctx.QueryArgs() + clientID := string(q.Peek("client_id")) + redirectURIRaw := string(q.Peek("redirect_uri")) + state := string(q.Peek("state")) + codeChallenge := string(q.Peek("code_challenge")) + codeChallengeMethod := string(q.Peek("code_challenge_method")) + resource := string(q.Peek("resource")) + scope := string(q.Peek("scope")) + + // Validate client exists before using redirect_uri. + client, err := h.store.ConfigStore.GetOAuth2ClientByClientID(ctx, clientID) + if err != nil || client == nil { + if errors.Is(err, configstore.ErrNotFound) { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_client", "unknown client_id") + return + } + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "failed to look up client") + return + } + + // Validate redirect_uri (loopback any-port per RFC 8252 §7.3). + if !matchRedirectURI(redirectURIRaw, client.RedirectURIs) { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_redirect_uri", "redirect_uri not registered for this client") + return + } + + // From here errors redirect to the client. + redirectError := func(errCode, description string) { + redirectWithParams(ctx, redirectURIRaw, map[string]string{ + "error": errCode, + "error_description": description, + "state": state, + }) + } + + // Constrain the requested scope to what the client registered for. scope is + // later copied into the access-token and refresh-token state, so a client must + // not be able to request a broader scope than it registered. An omitted scope + // defaults to the client's registered scope. + registeredScope := client.Scope + if registeredScope == "" { + registeredScope = "mcp" + } + if scope == "" { + scope = registeredScope + } else if !scopeWithinRegistered(scope, registeredScope) { + redirectError("invalid_scope", "requested scope exceeds the scope registered for this client") + return + } + + if string(q.Peek("response_type")) != "code" { + redirectError("unsupported_response_type", "only response_type=code is supported") + return + } + if codeChallengeMethod != "S256" { + redirectError("invalid_request", "code_challenge_method must be S256") + return + } + if codeChallenge == "" { + redirectError("invalid_request", "code_challenge is required") + return + } + // RFC 8707: bind the grant to this server's single protected resource. /mcp + // is the only resource we issue tokens for and token verification pins the + // audience to it. A client that omits resource (e.g. one that doesn't fetch + // the protected-resource metadata) defaults to the canonical /mcp resource + // since there is exactly one; a client that does send it must match. + canonicalResource := oauth2MCPResourceURL(ctx, h.store) + if resource == "" { + resource = canonicalResource + } else if resource != canonicalResource { + redirectError("invalid_target", "resource does not identify this MCP server") + return + } + + cfg := oauth2ServerCfg(h.store) + authCodeTTL := cfg.AuthCodeTTL + if authCodeTTL <= 0 { + authCodeTTL = configtables.DefaultAuthCodeTTL + } else if authCodeTTL > configtables.MaxAuthCodeTTL { + // Defense in depth: the API rejects an over-max value at save and config + // load rejects it at startup, so this should be unreachable — but clamp + // anyway so a code can never be minted with a lifetime above the cap. + authCodeTTL = configtables.MaxAuthCodeTTL + } + + req := &configtables.TableOAuth2AuthorizeRequest{ + ID: uuid.New().String(), + ClientID: client.ClientID, + RedirectURI: redirectURIRaw, + State: state, + Scope: scope, + Resource: resource, + CodeChallenge: codeChallenge, + CodeChallengeMethod: codeChallengeMethod, + Status: configtables.OAuth2AuthorizeRequestStatusPending, + ExpiresAt: time.Now().Add(time.Duration(authCodeTTL) * time.Second), + CreatedAt: time.Now(), + UpdatedAt: time.Now(), + } + if err := h.store.ConfigStore.CreateOAuth2AuthorizeRequest(ctx, req); err != nil { + // Keep the DB error server-side — it can carry table/constraint names that + // must not leak to the (untrusted) redirect target. + logger.Error("failed to create oauth2 authorize request: %v", err) + redirectError("server_error", "failed to create authorization request") + return + } + + // Mint a temp token scoping the consent page to this request. Without it the + // consent-page API calls (GET/PUT /api/oauth2/consent/flows/{id}) have no auth + // credential and fail with 401, leaving the user on a dead-end page — so a mint + // failure must abort with a well-formed error redirect rather than be swallowed. + tempToken := "" + if h.tempTokens != nil { + tok, err := h.tempTokens.Mint(ctx, temptoken.OAuth2ConsentScopeName, req.ID, time.Duration(authCodeTTL)*time.Second) + if err != nil { + logger.Error("failed to mint oauth2 consent temp token: %v", err) + redirectError("server_error", "failed to prepare consent flow") + return + } + tempToken = tok + } + + base := oauth2IssuerURL(ctx, h.store) + consentURL := fmt.Sprintf("%s/oauth/consent?flow=%s", base, url.QueryEscape(req.ID)) + if tempToken != "" { + consentURL += "#t=" + url.QueryEscape(tempToken) + } + + ctx.Response.Header.Set("Location", consentURL) + ctx.SetStatusCode(fasthttp.StatusFound) +} + +// --- POST /oauth2/token --- + +func (h *OAuth2IssuanceHandler) handleToken(ctx *fasthttp.RequestCtx) { + if h.store.ConfigStore == nil { + sendOAuthError(ctx, fasthttp.StatusServiceUnavailable, "server_error", "config store unavailable") + return + } + + grantType := string(ctx.FormValue("grant_type")) + switch grantType { + case "authorization_code": + h.handleTokenAuthCode(ctx) + case "refresh_token": + h.handleTokenRefresh(ctx) + default: + sendOAuthError(ctx, fasthttp.StatusBadRequest, "unsupported_grant_type", fmt.Sprintf("grant_type %q not supported", grantType)) + } +} + +func (h *OAuth2IssuanceHandler) handleTokenAuthCode(ctx *fasthttp.RequestCtx) { + code := string(ctx.FormValue("code")) + codeVerifier := string(ctx.FormValue("code_verifier")) + redirectURI := string(ctx.FormValue("redirect_uri")) + clientID := clientIDFromRequest(ctx) + resource := string(ctx.FormValue("resource")) + + if code == "" || codeVerifier == "" || clientID == "" { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_request", "code, code_verifier and client_id are required") + return + } + + // Look up the authorize request by hashing the received code. + codeHash := hashSHA256Hex(code) + req, err := h.store.ConfigStore.GetOAuth2AuthorizeRequestByCodeHash(ctx, codeHash) + if err != nil || req == nil { + if errors.Is(err, configstore.ErrNotFound) { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_grant", "authorization code not found or already used") + return + } + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "failed to look up authorization code") + return + } + if time.Now().After(req.ExpiresAt) { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_grant", "authorization code expired") + return + } + if req.ClientID != clientID { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_grant", "client_id mismatch") + return + } + // The authorization request always binds a redirect_uri (it is validated + // against the client's registered URIs at /oauth2/authorize), so per RFC 6749 + // §4.1.3 the token request must present it and it must match exactly. Accepting + // a missing redirect_uri would let a code be exchanged outside its bound redirect. + if redirectURI == "" || req.RedirectURI != redirectURI { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_grant", "redirect_uri mismatch") + return + } + if resource != "" && req.Resource != resource { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_grant", "resource mismatch") + return + } + + // Verify PKCE: SHA256(verifier) must equal stored challenge. + if !verifyPKCES256(codeVerifier, req.CodeChallenge) { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_grant", "PKCE verification failed") + return + } + + accessToken, refreshToken, refreshTokenObj, err := h.issueTokenPair(ctx, req.ID, req.ClientID, req.BfMode, req.BfSub, req.Scope, req.Resource) + if err != nil { + return + } + // Atomically mark the code as consumed and create the refresh token. + // If this fails the authorize request stays "consented" and the client can retry. + if err := h.store.ConfigStore.ConsumeOAuth2AuthorizeRequest(ctx, req.ID, refreshTokenObj); err != nil { + if errors.Is(err, configstore.ErrNotFound) { + // The code was concurrently consumed or expired between lookup and consume. + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_grant", "authorization code not found or already used") + return + } + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "failed to issue token") + return + } + sendTokenResponse(ctx, accessToken, refreshToken, req.Scope, oauth2ServerCfg(h.store).AccessTokenTTL) +} + +func (h *OAuth2IssuanceHandler) handleTokenRefresh(ctx *fasthttp.RequestCtx) { + refreshToken := string(ctx.FormValue("refresh_token")) + clientID := clientIDFromRequest(ctx) + resource := string(ctx.FormValue("resource")) + + if refreshToken == "" || clientID == "" { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_request", "refresh_token and client_id are required") + return + } + + tokenHash := hashSHA256Hex(refreshToken) + rt, err := h.store.ConfigStore.GetOAuth2RefreshTokenByHash(ctx, tokenHash) + if errors.Is(err, configstore.ErrNotFound) { + // Token not found in the active set — check if it was previously issued + // and revoked. A revoked token being re-presented indicates the token + // family may be compromised (stolen token used before rotation). Revoke + // all active tokens in the family to limit the damage (RFC 9700 §2.2.2). + // Fail closed: if the lookup or the family revocation errors, surface + // server_error rather than reporting invalid_grant while leaving a + // potentially compromised family usable. + revoked, lookupErr := h.store.ConfigStore.GetOAuth2RefreshTokenByHashAny(ctx, tokenHash) + switch { + case lookupErr != nil && !errors.Is(lookupErr, configstore.ErrNotFound): + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "failed to verify refresh token revocation state") + return + case revoked != nil: + if revokeErr := h.store.ConfigStore.RevokeOAuth2RefreshTokensByFamilyID(ctx, revoked.FamilyID); revokeErr != nil { + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "failed to revoke refresh token family") + return + } + } + // Unknown token or a revoked one we just contained — either way the grant + // is not usable. + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_grant", "refresh token not found or revoked") + return + } + if err != nil { + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "failed to look up refresh token") + return + } + if rt.ClientID != clientID { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_grant", "client_id mismatch") + return + } + + // VK identity cutoff (refresh side): when virtual-key identity has been + // disabled, vk-mode grants must not refresh. Live /mcp requests are already + // rejected at request time (see getMCPServerForRequest); denying refresh here + // closes the second path so a disabled grant can neither be used nor renewed. + // Gated on user identity being available, identical to the consent flow's + // availableModes — so this can never fire where vk is still the offered path. + if schemas.MCPAuthMode(rt.BfMode) == schemas.MCPAuthModeVK { + userModeAvailable := h.identityResolver != nil && h.identityResolver.IsUserModeAvailable() + if userModeAvailable && oauth2ServerCfg(h.store).DisableVKIdentity { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_grant", "virtual-key identity is no longer accepted; re-authenticate") + return + } + } + + // bf_sub liveness check: for VK-mode tokens, verify the VK still exists + // and is active. A deleted or disabled VK should not be able to silently + // obtain new access tokens via refresh. A transient lookup failure must stay + // retriable (server_error) — only a missing/inactive VK invalidates the grant. + if schemas.MCPAuthMode(rt.BfMode) == schemas.MCPAuthModeVK && h.store.ConfigStore != nil { + vk, vkErr := h.store.ConfigStore.GetVirtualKey(ctx, rt.BfSub) + if vkErr != nil && !errors.Is(vkErr, configstore.ErrNotFound) { + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "failed to verify virtual key") + return + } + if errors.Is(vkErr, configstore.ErrNotFound) || vk == nil || !vk.IsActiveValue() { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_grant", "virtual key is no longer active") + return + } + } + + // bf_sub liveness check: for user-mode tokens, verify the user still exists + // and is active, mirroring the VK-mode check above. A deleted or deactivated + // user should not be able to silently obtain new access tokens via refresh. + // The user is verified against the same in-memory source used on the /mcp + // request path. A transient lookup failure stays retriable (server_error) — + // only a gone/deactivated user invalidates the grant. + if schemas.MCPAuthMode(rt.BfMode) == schemas.MCPAuthModeUser && h.identityResolver != nil { + active, err := h.identityResolver.IsUserActive(ctx, rt.BfSub) + if err != nil { + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "failed to verify user") + return + } + if !active { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_grant", "user is no longer active") + return + } + } + + // RFC 8707: resource (audience URI) is distinct from scope. When the client + // omits it on refresh, carry forward the original resource captured at + // authorization — never substitute the scope string. When the client does + // provide it, it must match the resource bound at authorization: issuing a + // token for a different resource than originally authorized would let a + // refresh-token holder escape the original audience binding. Mirrors the + // auth-code handler's check. + if resource == "" { + resource = rt.Resource + } else if resource != rt.Resource { + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_grant", "resource mismatch") + return + } + + accessToken, newRefreshToken, newRefreshTokenObj, err := h.issueTokenPair(ctx, rt.FamilyID, rt.ClientID, rt.BfMode, rt.BfSub, rt.Scope, resource) + if err != nil { + return + } + // Carry the original grant's creation time forward across rotations so the + // grant's "Created" timestamp stays anchored to when it was first authorized, + // and stamp last_used_at to mark this refresh as the grant's latest activity. + usedAt := time.Now() + newRefreshTokenObj.CreatedAt = rt.CreatedAt + newRefreshTokenObj.LastUsedAt = &usedAt + // Atomically revoke the old token and create the new one. + // If this fails the old token stays active and the client can retry the refresh. + if err := h.store.ConfigStore.RotateOAuth2RefreshToken(ctx, rt.ID, newRefreshTokenObj); err != nil { + if errors.Is(err, configstore.ErrNotFound) { + // The token was concurrently rotated/revoked between lookup and rotate. + sendOAuthError(ctx, fasthttp.StatusBadRequest, "invalid_grant", "refresh token not found or revoked") + return + } + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "failed to rotate token") + return + } + sendTokenResponse(ctx, accessToken, newRefreshToken, rt.Scope, oauth2ServerCfg(h.store).AccessTokenTTL) +} + +// issueTokenPair mints a signed JWT access token and builds a refresh token row. +// It is a pure function — no DB writes. The caller is responsible for atomically +// persisting the refresh token row alongside any grant-specific side-effects +// (e.g. marking the auth code consumed, or revoking the previous refresh token). +// +// familyID traces the token back to its original authorization grant; all +// rotated descendants share the same ID for stolen-token detection (RFC 9700 §2.2.2). +// +// On error, issueTokenPair writes an OAuth error response to ctx and returns a +// non-nil error so the caller can return immediately without writing again. +func (h *OAuth2IssuanceHandler) issueTokenPair( + ctx *fasthttp.RequestCtx, + familyID, clientID, bfMode, bfSub, scope, resource string, +) (accessToken, refreshTokenPlain string, rt *configtables.TableOAuth2RefreshToken, err error) { + cfg := oauth2ServerCfg(h.store) + accessTokenTTL := cfg.AccessTokenTTL + if accessTokenTTL <= 0 { + accessTokenTTL = configtables.DefaultAccessTokenTTL + } + + signingKey, err := h.store.GetOAuth2SigningKey(ctx) + if err != nil { + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "signing key unavailable") + return + } + privKey, err := parseRSAPrivateKeyPEM(signingKey.PrivateKeyPEM) + if err != nil { + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "invalid signing key") + return + } + + issuer := oauth2IssuerURL(ctx, h.store) + now := time.Now() + claims := jwt.MapClaims{ + "iss": issuer, + "aud": jwt.ClaimStrings{resource}, + "sub": bfSub, + "bf_mode": bfMode, + "scope": scope, + "iat": now.Unix(), + "nbf": now.Unix(), + "exp": now.Add(time.Duration(accessTokenTTL) * time.Second).Unix(), + } + tok := jwt.NewWithClaims(jwt.SigningMethodRS256, claims) + tok.Header["kid"] = signingKey.KID + accessToken, err = tok.SignedString(privKey) + if err != nil { + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "failed to sign access token") + return + } + + refreshTokenPlain, err = generateSecureToken(32) + if err != nil { + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "failed to generate refresh token") + return + } + rt = &configtables.TableOAuth2RefreshToken{ + ID: uuid.New().String(), + TokenHash: hashSHA256Hex(refreshTokenPlain), + FamilyID: familyID, + ClientID: clientID, + BfMode: bfMode, + BfSub: bfSub, + Scope: scope, + Resource: resource, + CreatedAt: now, + } + return +} + +// sendTokenResponse writes the RFC 6749 token response to ctx. +func sendTokenResponse(ctx *fasthttp.RequestCtx, accessToken, refreshToken, scope string, accessTokenTTL int) { + if accessTokenTTL <= 0 { + accessTokenTTL = configtables.DefaultAccessTokenTTL + } + ctx.SetStatusCode(fasthttp.StatusOK) + ctx.SetContentType("application/json") + ctx.Response.Header.Set("Cache-Control", "no-store") + ctx.Response.Header.Set("Pragma", "no-cache") + data, err := sonic.Marshal(map[string]any{ + "access_token": accessToken, + "token_type": "Bearer", + "expires_in": accessTokenTTL, + "refresh_token": refreshToken, + "scope": scope, + }) + if err != nil { + sendOAuthError(ctx, fasthttp.StatusInternalServerError, "server_error", "failed to marshal token response") + return + } + ctx.SetBody(data) +} + +// clientIDFromRequest resolves the OAuth client_id for a token request. Public +// clients (token_endpoint_auth_method=none) may send it either as a request +// parameter or as the username of an HTTP Basic Authorization header +// (RFC 6749 §2.3.1); some clients default to the header form. Both are accepted; +// the Basic password is ignored since only public clients are supported. +func clientIDFromRequest(ctx *fasthttp.RequestCtx) string { + if v := string(ctx.FormValue("client_id")); v != "" { + return v + } + auth := string(ctx.Request.Header.Peek("Authorization")) + const prefix = "Basic " + if len(auth) > len(prefix) && strings.EqualFold(auth[:len(prefix)], prefix) { + if decoded, err := base64.StdEncoding.DecodeString(strings.TrimSpace(auth[len(prefix):])); err == nil { + // Basic credentials are "client_id:client_secret", each + // application/x-www-form-urlencoded; take and decode the username. + username, _, _ := strings.Cut(string(decoded), ":") + if id, uerr := url.QueryUnescape(username); uerr == nil { + return id + } + return username + } + } + return "" +} + +// --- Helpers --- + +// isLoopbackRedirectHost reports whether a redirect URI host is a loopback +// address per RFC 8252 §7.3: localhost, 127.0.0.1, or the IPv6 loopback [::1] +// (url.Hostname() already strips the brackets). +func isLoopbackRedirectHost(host string) bool { + if host == "localhost" { + return true + } + ip := net.ParseIP(host) + return ip != nil && ip.IsLoopback() +} + +// matchRedirectURI validates redirect_uri against registered URIs. +// For loopback addresses (localhost / 127.0.0.1 / [::1]), port is ignored per RFC 8252 §7.3. +// isAllowedRedirectScheme reports whether a redirect URI uses a safe scheme: +// https for any host, or http only for loopback addresses (localhost/127.0.0.1/[::1]). +// This rejects javascript:, data:, and other schemes that could be abused when a +// Location header is built from the URI. +func isAllowedRedirectScheme(candidate string) bool { + parsed, err := url.Parse(candidate) + if err != nil { + return false + } + switch parsed.Scheme { + case "https": + return true + case "http": + return isLoopbackRedirectHost(parsed.Hostname()) + default: + return false + } +} + +func matchRedirectURI(candidate string, registered []string) bool { + parsed, err := url.Parse(candidate) + if err != nil { + return false + } + isLoopback := isLoopbackRedirectHost(parsed.Hostname()) + + for _, r := range registered { + rParsed, err := url.Parse(r) + if err != nil { + continue + } + if isLoopback && isLoopbackRedirectHost(rParsed.Hostname()) { + // Loopback: match scheme + host (without port) + path. + if parsed.Scheme == rParsed.Scheme && parsed.Path == rParsed.Path { + return true + } + } else { + // Non-loopback: exact match. + if candidate == r { + return true + } + } + } + return false +} + +// redirectWithParams builds a redirect URL with the given query params and redirects. +func redirectWithParams(ctx *fasthttp.RequestCtx, base string, params map[string]string) { + u, err := url.Parse(base) + if err != nil { + ctx.SetStatusCode(fasthttp.StatusInternalServerError) + return + } + q := u.Query() + for k, v := range params { + if v != "" { + q.Set(k, v) + } + } + u.RawQuery = q.Encode() + ctx.Response.Header.Set("Location", u.String()) + ctx.SetStatusCode(fasthttp.StatusFound) +} + +// scopeWithinRegistered reports whether every space-delimited token in requested +// is present in the registered scope set. Callers default an empty requested scope +// to the registered scope before calling, so an empty requested scope is vacuously +// within bounds. +func scopeWithinRegistered(requested, registered string) bool { + allowed := make(map[string]struct{}) + for s := range strings.FieldsSeq(registered) { + allowed[s] = struct{}{} + } + for s := range strings.FieldsSeq(requested) { + if _, ok := allowed[s]; !ok { + return false + } + } + return true +} + +// verifyPKCES256 verifies a PKCE S256 code_verifier against a stored challenge. +func verifyPKCES256(verifier, challenge string) bool { + h := sha256.Sum256([]byte(verifier)) + computed := base64.RawURLEncoding.EncodeToString(h[:]) + return subtle.ConstantTimeCompare([]byte(computed), []byte(challenge)) == 1 +} + +// hashSHA256Hex returns the hex-encoded SHA-256 hash of the input. +func hashSHA256Hex(input string) string { + h := sha256.Sum256([]byte(input)) + return hex.EncodeToString(h[:]) +} + +// generateSecureToken returns a cryptographically secure URL-safe random token. +func generateSecureToken(length int) (string, error) { + b := make([]byte, length) + if _, err := rand.Read(b); err != nil { + return "", err + } + return base64.RawURLEncoding.EncodeToString(b), nil +} + +// parseRSAPrivateKeyPEM decodes and parses a PKCS8 RSA private key PEM. +func parseRSAPrivateKeyPEM(pemStr string) (*rsa.PrivateKey, error) { + block, rest := pem.Decode([]byte(pemStr)) + if block == nil || len(rest) > 0 { + return nil, fmt.Errorf("malformed private key PEM") + } + key, err := x509.ParsePKCS8PrivateKey(block.Bytes) + if err != nil { + return nil, fmt.Errorf("parse private key: %w", err) + } + rsaKey, ok := key.(*rsa.PrivateKey) + if !ok { + return nil, fmt.Errorf("expected RSA private key, got %T", key) + } + return rsaKey, nil +} + +// sendOAuthError writes an RFC 6749 §5.2 error response. +func sendOAuthError(ctx *fasthttp.RequestCtx, statusCode int, errCode, description string) { + ctx.SetStatusCode(statusCode) + ctx.SetContentType("application/json") + data, err := sonic.Marshal(map[string]string{ + "error": errCode, + "error_description": description, + }) + if err != nil { + ctx.Error("failed to marshal error response", fasthttp.StatusInternalServerError) + return + } + ctx.SetBody(data) +} diff --git a/transports/bifrost-http/handlers/mcpoauth2issuance_test.go b/transports/bifrost-http/handlers/mcpoauth2issuance_test.go new file mode 100644 index 00000000000..51763e1787e --- /dev/null +++ b/transports/bifrost-http/handlers/mcpoauth2issuance_test.go @@ -0,0 +1,707 @@ +package handlers + +import ( + "context" + "crypto/sha256" + "encoding/base64" + "encoding/json" + "net/url" + "path/filepath" + "testing" + "time" + + "github.com/maximhq/bifrost/framework/configstore" + configtables "github.com/maximhq/bifrost/framework/configstore/tables" + "github.com/maximhq/bifrost/transports/bifrost-http/lib" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "github.com/valyala/fasthttp" +) + +// newRealOAuth2Store builds a real sqlite-backed ConfigStore (full migrations, +// including the OAuth2 issuance tables) so issuance handlers exercise the actual +// atomic store semantics rather than a hand-rolled mock. +func newRealOAuth2Store(t *testing.T) configstore.ConfigStore { + t.Helper() + cs, err := configstore.NewConfigStore(context.Background(), &configstore.Config{ + Enabled: true, + Type: configstore.ConfigStoreTypeSQLite, + Config: &configstore.SQLiteConfig{Path: filepath.Join(t.TempDir(), "oauth2.db")}, + }, &mockLogger{}) + require.NoError(t, err) + require.NotNil(t, cs) + return cs +} + +func newIssuanceHandler(t *testing.T) (*OAuth2IssuanceHandler, configstore.ConfigStore, *lib.Config) { + t.Helper() + SetLogger(&mockLogger{}) + store := newRealOAuth2Store(t) + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + return NewOAuth2IssuanceHandler(cfg, nil, nil), store, cfg +} + +func pkceChallenge(verifier string) string { + sum := sha256.Sum256([]byte(verifier)) + return base64.RawURLEncoding.EncodeToString(sum[:]) +} + +// initCtx prepares a RequestCtx for handler use. Init sets a (fake) server so +// that, when the handler passes ctx to the store as a context.Context, the +// database/sql layer's ctx.Done() call does not panic on a bare RequestCtx. +func initCtx(req *fasthttp.Request) *fasthttp.RequestCtx { + ctx := &fasthttp.RequestCtx{} + ctx.Init(req, nil, nil) + return ctx +} + +// bgCtx returns an initialized, empty RequestCtx for store-backed calls made +// directly from a test (e.g. verifying a minted token against the live store). +func bgCtx() *fasthttp.RequestCtx { + var req fasthttp.Request + return initCtx(&req) +} + +func formPostCtx(body string) *fasthttp.RequestCtx { + var req fasthttp.Request + req.Header.SetMethod("POST") + req.Header.SetContentType("application/x-www-form-urlencoded") + req.SetBodyString(body) + return initCtx(&req) +} + +func getCtx(uri string) *fasthttp.RequestCtx { + var req fasthttp.Request + req.Header.SetMethod("GET") + req.SetRequestURI(uri) + return initCtx(&req) +} + +// seedClient registers a client directly via the store and returns its client_id. +func seedClient(t *testing.T, store configstore.ConfigStore, redirectURIs []string) string { + t.Helper() + client := &configtables.TableOAuth2Client{ + ID: "client-row-1", + ClientID: "client-1", + ClientName: "Test Client", + RedirectURIs: redirectURIs, + GrantTypes: []string{"authorization_code"}, + Scope: "mcp", + CreatedAt: time.Now(), + } + require.NoError(t, store.CreateOAuth2Client(context.Background(), client)) + return client.ClientID +} + +// seedConsentedRequest stores a consented authorize request bound to the given +// identity, with a code hash and PKCE challenge, returning the plaintext code. +func seedConsentedRequest(t *testing.T, store configstore.ConfigStore, id, clientID, code, challenge, bfMode, bfSub string, expires time.Time) { + t.Helper() + h := hashSHA256Hex(code) + req := &configtables.TableOAuth2AuthorizeRequest{ + ID: id, + ClientID: clientID, + RedirectURI: "http://127.0.0.1/cb", + State: "state", + Scope: "mcp", + Resource: testMCPResource, + CodeChallenge: challenge, + CodeChallengeMethod: "S256", + Status: configtables.OAuth2AuthorizeRequestStatusConsented, + BfMode: bfMode, + BfSub: bfSub, + CodeHash: &h, + ExpiresAt: expires, + CreatedAt: time.Now(), + UpdatedAt: time.Now(), + } + require.NoError(t, store.CreateOAuth2AuthorizeRequest(context.Background(), req)) +} + +func TestHandleRegister_DCR(t *testing.T) { + t.Run("valid registration returns 201 with defaults", func(t *testing.T) { + h, _, _ := newIssuanceHandler(t) + ctx := formPostCtx("") + ctx.Request.SetBodyString(`{"client_name":"Cli","redirect_uris":["http://127.0.0.1:1234/cb"]}`) + ctx.Request.Header.SetContentType("application/json") + + h.handleRegister(ctx) + require.Equal(t, fasthttp.StatusCreated, ctx.Response.StatusCode()) + + var resp map[string]any + require.NoError(t, json.Unmarshal(ctx.Response.Body(), &resp)) + assert.NotEmpty(t, resp["client_id"]) + assert.Equal(t, "none", resp["token_endpoint_auth_method"]) + assert.Equal(t, "mcp", resp["scope"]) + assert.Equal(t, []any{"authorization_code"}, resp["grant_types"]) + }) + + t.Run("missing redirect_uris is rejected", func(t *testing.T) { + h, _, _ := newIssuanceHandler(t) + ctx := formPostCtx("") + ctx.Request.SetBodyString(`{"client_name":"Cli"}`) + ctx.Request.Header.SetContentType("application/json") + + h.handleRegister(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "invalid_redirect_uri") + }) + + t.Run("non-public auth method is rejected", func(t *testing.T) { + h, _, _ := newIssuanceHandler(t) + ctx := formPostCtx("") + ctx.Request.SetBodyString(`{"redirect_uris":["http://127.0.0.1/cb"],"token_endpoint_auth_method":"client_secret_basic"}`) + ctx.Request.Header.SetContentType("application/json") + + h.handleRegister(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "invalid_client_metadata") + }) + + t.Run("malformed JSON is rejected", func(t *testing.T) { + h, _, _ := newIssuanceHandler(t) + ctx := formPostCtx("") + ctx.Request.SetBodyString(`{not json`) + ctx.Request.Header.SetContentType("application/json") + + h.handleRegister(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "invalid_request") + }) + + t.Run("unsupported grant_type is rejected", func(t *testing.T) { + h, _, _ := newIssuanceHandler(t) + ctx := formPostCtx("") + ctx.Request.SetBodyString(`{"redirect_uris":["http://127.0.0.1/cb"],"grant_types":["client_credentials"]}`) + ctx.Request.Header.SetContentType("application/json") + + h.handleRegister(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "invalid_client_metadata") + }) + + t.Run("unsupported response_type is rejected", func(t *testing.T) { + h, _, _ := newIssuanceHandler(t) + ctx := formPostCtx("") + ctx.Request.SetBodyString(`{"redirect_uris":["http://127.0.0.1/cb"],"response_types":["token"]}`) + ctx.Request.Header.SetContentType("application/json") + + h.handleRegister(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "invalid_client_metadata") + }) +} + +func TestHandleAuthorize(t *testing.T) { + base := func(clientID, redirect string) url.Values { + v := url.Values{} + v.Set("client_id", clientID) + v.Set("redirect_uri", redirect) + v.Set("response_type", "code") + v.Set("code_challenge", "challenge") + v.Set("code_challenge_method", "S256") + v.Set("resource", testMCPResource) + v.Set("state", "xyz") + return v + } + + t.Run("happy path redirects to the consent page", func(t *testing.T) { + h, store, _ := newIssuanceHandler(t) + cid := seedClient(t, store, []string{"http://127.0.0.1:1234/cb"}) + ctx := getCtx("/oauth2/authorize?" + base(cid, "http://127.0.0.1:1234/cb").Encode()) + + h.handleAuthorize(ctx) + require.Equal(t, fasthttp.StatusFound, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Header.Peek("Location")), "/oauth/consent?flow=") + }) + + t.Run("loopback redirect matches on any port", func(t *testing.T) { + h, store, _ := newIssuanceHandler(t) + cid := seedClient(t, store, []string{"http://127.0.0.1:1234/cb"}) + // Registered port 1234, request uses 55555 — must still match (RFC 8252). + ctx := getCtx("/oauth2/authorize?" + base(cid, "http://127.0.0.1:55555/cb").Encode()) + + h.handleAuthorize(ctx) + assert.Equal(t, fasthttp.StatusFound, ctx.Response.StatusCode()) + }) + + t.Run("unknown client_id is rejected", func(t *testing.T) { + h, _, _ := newIssuanceHandler(t) + ctx := getCtx("/oauth2/authorize?" + base("nope", "http://127.0.0.1/cb").Encode()) + + h.handleAuthorize(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "invalid_client") + }) + + t.Run("unregistered redirect_uri is rejected", func(t *testing.T) { + h, store, _ := newIssuanceHandler(t) + cid := seedClient(t, store, []string{"http://127.0.0.1/cb"}) + ctx := getCtx("/oauth2/authorize?" + base(cid, "https://evil.example/cb").Encode()) + + h.handleAuthorize(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "invalid_redirect_uri") + }) + + // Once client + redirect validate, protocol errors redirect back to the client. + redirectCases := []struct { + name string + mutate func(url.Values) + errCode string + }{ + {"non-code response_type", func(v url.Values) { v.Set("response_type", "token") }, "unsupported_response_type"}, + {"non-S256 challenge method", func(v url.Values) { v.Set("code_challenge_method", "plain") }, "invalid_request"}, + {"mismatched resource", func(v url.Values) { v.Set("resource", "https://evil.example/mcp") }, "invalid_target"}, + {"scope exceeds registered", func(v url.Values) { v.Set("scope", "mcp admin") }, "invalid_scope"}, + } + for _, tc := range redirectCases { + t.Run(tc.name+" redirects with error", func(t *testing.T) { + h, store, _ := newIssuanceHandler(t) + cid := seedClient(t, store, []string{"http://127.0.0.1/cb"}) + v := base(cid, "http://127.0.0.1/cb") + tc.mutate(v) + ctx := getCtx("/oauth2/authorize?" + v.Encode()) + + h.handleAuthorize(ctx) + require.Equal(t, fasthttp.StatusFound, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Header.Peek("Location")), "error="+tc.errCode) + }) + } + + // RFC 8707: this server exposes exactly one protected resource (/mcp), so a + // client that omits resource defaults to the canonical one and proceeds. + t.Run("omitted resource defaults to canonical and proceeds", func(t *testing.T) { + h, store, _ := newIssuanceHandler(t) + cid := seedClient(t, store, []string{"http://127.0.0.1/cb"}) + v := base(cid, "http://127.0.0.1/cb") + v.Del("resource") + ctx := getCtx("/oauth2/authorize?" + v.Encode()) + + h.handleAuthorize(ctx) + require.Equal(t, fasthttp.StatusFound, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Header.Peek("Location")), "/oauth/consent?flow=") + }) +} + +func TestHandleToken_AuthorizationCode(t *testing.T) { + const verifier = "test-verifier-string-of-sufficient-length-1234567890" + challenge := pkceChallenge(verifier) + + t.Run("happy path issues a verifiable token pair", func(t *testing.T) { + h, store, cfg := newIssuanceHandler(t) + cid := seedClient(t, store, []string{"http://127.0.0.1/cb"}) + seedConsentedRequest(t, store, "req-1", cid, "code-1", challenge, "session", "sess-1", time.Now().Add(time.Minute)) + + body := url.Values{ + "grant_type": {"authorization_code"}, + "code": {"code-1"}, + "code_verifier": {verifier}, + "client_id": {cid}, + "redirect_uri": {"http://127.0.0.1/cb"}, + }.Encode() + ctx := formPostCtx(body) + h.handleToken(ctx) + require.Equal(t, fasthttp.StatusOK, ctx.Response.StatusCode(), string(ctx.Response.Body())) + + var resp map[string]any + require.NoError(t, json.Unmarshal(ctx.Response.Body(), &resp)) + assert.Equal(t, "Bearer", resp["token_type"]) + assert.NotEmpty(t, resp["refresh_token"]) + + // The minted access token verifies under the same issuer/signing key. + signingKey, keyErr := cfg.ConfigStore.GetOAuth2SigningKey(bgCtx()) + require.NoError(t, keyErr) + claims, err := verifyMCPJWT(bgCtx(), resp["access_token"].(string), cfg, signingKey) + require.NoError(t, err) + assert.Equal(t, "session", claims.BfMode) + assert.Equal(t, "sess-1", claims.Subject) + }) + + t.Run("PKCE mismatch is rejected", func(t *testing.T) { + h, store, _ := newIssuanceHandler(t) + cid := seedClient(t, store, []string{"http://127.0.0.1/cb"}) + seedConsentedRequest(t, store, "req-1", cid, "code-1", challenge, "session", "sess-1", time.Now().Add(time.Minute)) + + ctx := formPostCtx(url.Values{ + "grant_type": {"authorization_code"}, + "code": {"code-1"}, + "code_verifier": {"wrong-verifier"}, + "client_id": {cid}, + "redirect_uri": {"http://127.0.0.1/cb"}, + }.Encode()) + h.handleToken(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "invalid_grant") + }) + + t.Run("code is single-use", func(t *testing.T) { + h, store, _ := newIssuanceHandler(t) + cid := seedClient(t, store, []string{"http://127.0.0.1/cb"}) + seedConsentedRequest(t, store, "req-1", cid, "code-1", challenge, "session", "sess-1", time.Now().Add(time.Minute)) + + body := url.Values{ + "grant_type": {"authorization_code"}, + "code": {"code-1"}, + "code_verifier": {verifier}, + "client_id": {cid}, + "redirect_uri": {"http://127.0.0.1/cb"}, + }.Encode() + first := formPostCtx(body) + h.handleToken(first) + require.Equal(t, fasthttp.StatusOK, first.Response.StatusCode()) + + second := formPostCtx(body) + h.handleToken(second) + assert.Equal(t, fasthttp.StatusBadRequest, second.Response.StatusCode()) + assert.Contains(t, string(second.Response.Body()), "invalid_grant") + }) + + t.Run("expired code is rejected", func(t *testing.T) { + h, store, _ := newIssuanceHandler(t) + cid := seedClient(t, store, []string{"http://127.0.0.1/cb"}) + seedConsentedRequest(t, store, "req-1", cid, "code-1", challenge, "session", "sess-1", time.Now().Add(-time.Minute)) + + ctx := formPostCtx(url.Values{ + "grant_type": {"authorization_code"}, + "code": {"code-1"}, + "code_verifier": {verifier}, + "client_id": {cid}, + "redirect_uri": {"http://127.0.0.1/cb"}, + }.Encode()) + h.handleToken(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "invalid_grant") + }) + + t.Run("client_id mismatch is rejected", func(t *testing.T) { + h, store, _ := newIssuanceHandler(t) + cid := seedClient(t, store, []string{"http://127.0.0.1/cb"}) + seedConsentedRequest(t, store, "req-1", cid, "code-1", challenge, "session", "sess-1", time.Now().Add(time.Minute)) + + ctx := formPostCtx(url.Values{ + "grant_type": {"authorization_code"}, + "code": {"code-1"}, + "code_verifier": {verifier}, + "client_id": {"other-client"}, + "redirect_uri": {"http://127.0.0.1/cb"}, + }.Encode()) + h.handleToken(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "invalid_grant") + }) + + t.Run("missing redirect_uri is rejected", func(t *testing.T) { + h, store, _ := newIssuanceHandler(t) + cid := seedClient(t, store, []string{"http://127.0.0.1/cb"}) + seedConsentedRequest(t, store, "req-1", cid, "code-1", challenge, "session", "sess-1", time.Now().Add(time.Minute)) + + ctx := formPostCtx(url.Values{ + "grant_type": {"authorization_code"}, + "code": {"code-1"}, + "code_verifier": {verifier}, + "client_id": {cid}, + }.Encode()) + h.handleToken(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "invalid_grant") + }) + + t.Run("missing required fields are rejected", func(t *testing.T) { + h, _, _ := newIssuanceHandler(t) + ctx := formPostCtx(url.Values{"grant_type": {"authorization_code"}}.Encode()) + h.handleToken(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "invalid_request") + }) + + t.Run("unsupported grant_type is rejected", func(t *testing.T) { + h, _, _ := newIssuanceHandler(t) + ctx := formPostCtx(url.Values{"grant_type": {"password"}}.Encode()) + h.handleToken(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "unsupported_grant_type") + }) +} + +func TestHandleToken_RefreshRotationAndReplay(t *testing.T) { + // Seed an initial active refresh token via the real consume path so the row + // is created exactly as issuance would. + seedRefresh := func(t *testing.T, store configstore.ConfigStore, plain string) { + t.Helper() + seedConsentedRequest(t, store, "fam-1", "client-1", "auth-code", pkceChallenge("v"), "session", "sess-1", time.Now().Add(time.Minute)) + rt := &configtables.TableOAuth2RefreshToken{ + ID: "rt-1", TokenHash: hashSHA256Hex(plain), FamilyID: "fam-1", ClientID: "client-1", + BfMode: "session", BfSub: "sess-1", Scope: "mcp", Resource: testMCPResource, CreatedAt: time.Now(), + } + require.NoError(t, store.ConsumeOAuth2AuthorizeRequest(context.Background(), "fam-1", rt)) + } + + t.Run("rotation issues a new pair and carries the family", func(t *testing.T) { + h, store, _ := newIssuanceHandler(t) + seedClient(t, store, []string{"http://127.0.0.1/cb"}) + seedRefresh(t, store, "refresh-plain-1") + + ctx := formPostCtx(url.Values{ + "grant_type": {"refresh_token"}, + "refresh_token": {"refresh-plain-1"}, + "client_id": {"client-1"}, + }.Encode()) + h.handleToken(ctx) + require.Equal(t, fasthttp.StatusOK, ctx.Response.StatusCode(), string(ctx.Response.Body())) + + var resp map[string]any + require.NoError(t, json.Unmarshal(ctx.Response.Body(), &resp)) + newRefresh := resp["refresh_token"].(string) + assert.NotEqual(t, "refresh-plain-1", newRefresh) + + // The new token is active under the same family. + active, err := store.GetOAuth2RefreshTokenByHash(context.Background(), hashSHA256Hex(newRefresh)) + require.NoError(t, err) + assert.Equal(t, "fam-1", active.FamilyID) + // The old token is no longer active. + _, err = store.GetOAuth2RefreshTokenByHash(context.Background(), hashSHA256Hex("refresh-plain-1")) + assert.ErrorIs(t, err, configstore.ErrNotFound) + }) + + t.Run("replaying a rotated token revokes the whole family", func(t *testing.T) { + h, store, _ := newIssuanceHandler(t) + seedClient(t, store, []string{"http://127.0.0.1/cb"}) + seedRefresh(t, store, "refresh-plain-1") + + rotate := formPostCtx(url.Values{ + "grant_type": {"refresh_token"}, "refresh_token": {"refresh-plain-1"}, "client_id": {"client-1"}, + }.Encode()) + h.handleToken(rotate) + require.Equal(t, fasthttp.StatusOK, rotate.Response.StatusCode()) + var resp map[string]any + require.NoError(t, json.Unmarshal(rotate.Response.Body(), &resp)) + newRefresh := resp["refresh_token"].(string) + + // Replay the now-revoked original. + replay := formPostCtx(url.Values{ + "grant_type": {"refresh_token"}, "refresh_token": {"refresh-plain-1"}, "client_id": {"client-1"}, + }.Encode()) + h.handleToken(replay) + assert.Equal(t, fasthttp.StatusBadRequest, replay.Response.StatusCode()) + assert.Contains(t, string(replay.Response.Body()), "invalid_grant") + + // The family is now fully revoked: the freshly issued token no longer works. + after := formPostCtx(url.Values{ + "grant_type": {"refresh_token"}, "refresh_token": {newRefresh}, "client_id": {"client-1"}, + }.Encode()) + h.handleToken(after) + assert.Equal(t, fasthttp.StatusBadRequest, after.Response.StatusCode()) + assert.Contains(t, string(after.Response.Body()), "invalid_grant") + }) + + t.Run("client_id mismatch is rejected", func(t *testing.T) { + h, store, _ := newIssuanceHandler(t) + seedClient(t, store, []string{"http://127.0.0.1/cb"}) + seedRefresh(t, store, "refresh-plain-1") + + ctx := formPostCtx(url.Values{ + "grant_type": {"refresh_token"}, "refresh_token": {"refresh-plain-1"}, "client_id": {"other"}, + }.Encode()) + h.handleToken(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "invalid_grant") + }) + + t.Run("missing fields are rejected", func(t *testing.T) { + h, _, _ := newIssuanceHandler(t) + ctx := formPostCtx(url.Values{"grant_type": {"refresh_token"}}.Encode()) + h.handleToken(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "invalid_request") + }) +} + +func TestHandleToken_RefreshVKIdentityDisabled(t *testing.T) { + seedVKRefresh := func(t *testing.T, store configstore.ConfigStore, plain string) { + t.Helper() + seedConsentedRequest(t, store, "fam-vk", "client-1", "auth-code-vk", pkceChallenge("v"), "vk", "vk-1", time.Now().Add(time.Minute)) + rt := &configtables.TableOAuth2RefreshToken{ + ID: "rt-vk", TokenHash: hashSHA256Hex(plain), FamilyID: "fam-vk", ClientID: "client-1", + BfMode: "vk", BfSub: "vk-1", Scope: "mcp", Resource: testMCPResource, CreatedAt: time.Now(), + } + require.NoError(t, store.ConsumeOAuth2AuthorizeRequest(context.Background(), "fam-vk", rt)) + } + + refreshReq := func() *fasthttp.RequestCtx { + return formPostCtx(url.Values{ + "grant_type": {"refresh_token"}, "refresh_token": {"refresh-vk-1"}, "client_id": {"client-1"}, + }.Encode()) + } + + t.Run("vk refresh rejected when disabled and user mode available", func(t *testing.T) { + store := newRealOAuth2Store(t) + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeOAuth, false) + cfg.ClientConfig.OAuth2ServerConfig.DisableVKIdentity = true + h := NewOAuth2IssuanceHandler(cfg, nil, &fakeResolver{userModeAvailable: true}) + seedClient(t, store, []string{"http://127.0.0.1/cb"}) + seedVKRefresh(t, store, "refresh-vk-1") + + ctx := refreshReq() + h.handleToken(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + body := string(ctx.Response.Body()) + assert.Contains(t, body, "invalid_grant") + assert.Contains(t, body, "no longer accepted") + }) + + t.Run("vk refresh not blocked by flag when user mode unavailable", func(t *testing.T) { + store := newRealOAuth2Store(t) + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeOAuth, false) + cfg.ClientConfig.OAuth2ServerConfig.DisableVKIdentity = true + // nil resolver → user mode unavailable → the flag is ignored and the flow + // falls through to the VK liveness check, which rejects the (unseeded) VK + // with a different message. That proves the cutoff did not fire. + h := NewOAuth2IssuanceHandler(cfg, nil, nil) + seedClient(t, store, []string{"http://127.0.0.1/cb"}) + seedVKRefresh(t, store, "refresh-vk-1") + + ctx := refreshReq() + h.handleToken(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + body := string(ctx.Response.Body()) + assert.NotContains(t, body, "no longer accepted") + assert.Contains(t, body, "no longer active") + }) +} + +func TestHandleToken_RefreshUserLiveness(t *testing.T) { + seedUserRefresh := func(t *testing.T, store configstore.ConfigStore, plain string) { + t.Helper() + seedConsentedRequest(t, store, "fam-user", "client-1", "auth-code-user", pkceChallenge("v"), "user", "user-1", time.Now().Add(time.Minute)) + rt := &configtables.TableOAuth2RefreshToken{ + ID: "rt-user", TokenHash: hashSHA256Hex(plain), FamilyID: "fam-user", ClientID: "client-1", + BfMode: "user", BfSub: "user-1", Scope: "mcp", Resource: testMCPResource, CreatedAt: time.Now(), + } + require.NoError(t, store.ConsumeOAuth2AuthorizeRequest(context.Background(), "fam-user", rt)) + } + + refreshReq := func() *fasthttp.RequestCtx { + return formPostCtx(url.Values{ + "grant_type": {"refresh_token"}, "refresh_token": {"refresh-user-1"}, "client_id": {"client-1"}, + }.Encode()) + } + + t.Run("user refresh rejected when the user is no longer active", func(t *testing.T) { + store := newRealOAuth2Store(t) + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeOAuth, false) + // Resolver reports the user as gone: refresh must be denied rather than + // minting a new access token, mirroring the VK-mode liveness check. + h := NewOAuth2IssuanceHandler(cfg, nil, &fakeResolver{userModeAvailable: true, userInactive: true}) + seedClient(t, store, []string{"http://127.0.0.1/cb"}) + seedUserRefresh(t, store, "refresh-user-1") + + ctx := refreshReq() + h.handleToken(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + body := string(ctx.Response.Body()) + assert.Contains(t, body, "invalid_grant") + assert.Contains(t, body, "user is no longer active") + }) + + t.Run("user refresh succeeds when the user is active", func(t *testing.T) { + store := newRealOAuth2Store(t) + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeOAuth, false) + // Active user (default resolver) must pass the liveness check and rotate. + h := NewOAuth2IssuanceHandler(cfg, nil, &fakeResolver{userModeAvailable: true}) + seedClient(t, store, []string{"http://127.0.0.1/cb"}) + seedUserRefresh(t, store, "refresh-user-1") + + ctx := refreshReq() + h.handleToken(ctx) + require.Equal(t, fasthttp.StatusOK, ctx.Response.StatusCode(), string(ctx.Response.Body())) + assert.NotContains(t, string(ctx.Response.Body()), "user is no longer active") + }) +} + +// putConfigCtx builds an initialized PUT /api/config request carrying a JSON body. +func putConfigCtx(body string) *fasthttp.RequestCtx { + var req fasthttp.Request + req.Header.SetMethod("PUT") + req.Header.SetContentType("application/json") + req.SetBodyString(body) + return initCtx(&req) +} + +// TestUpdateConfig_RejectsAuthCodeTTLAboveMax covers the API-layer guard: a save +// with auth_code_ttl above the cap is rejected with 400 in every mode — including +// headers, so the API can never persist a value the load-time validator would +// later reject at boot — before any live runtime mutation. The handler returns at +// the validation, so configManager is never invoked (left nil). +func TestUpdateConfig_RejectsAuthCodeTTLAboveMax(t *testing.T) { + for _, mode := range []configtables.MCPServerAuthMode{ + configtables.MCPServerAuthModeOAuth, + configtables.MCPServerAuthModeBoth, + configtables.MCPServerAuthModeHeaders, + } { + t.Run(string(mode), func(t *testing.T) { + SetLogger(&mockLogger{}) + store := newRealOAuth2Store(t) + cfg := newTestOAuth2Config(store, mode, false) + h := &ConfigHandler{store: cfg} + + body := `{"client_config":{"mcp_server_auth_mode":"` + string(mode) + + `","oauth2_server_config":{"auth_code_ttl":5000,"access_token_ttl":600}}}` + ctx := putConfigCtx(body) + h.updateConfig(ctx) + + require.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + assert.Contains(t, string(ctx.Response.Body()), "auth_code_ttl must not exceed") + }) + } +} + +// TestHandleAuthorize_AuthCodeTTLResolution covers the issuance-layer resolution +// of the configured auth_code_ttl into the authorization code's ExpiresAt: +// over-cap is clamped to the max, zero falls back to the default, and an in-range +// value is used verbatim. The three expected values are far enough apart that the +// generous timing tolerance cannot mask the wrong branch. +func TestHandleAuthorize_AuthCodeTTLResolution(t *testing.T) { + cases := []struct { + name string + configured int + wantTTL int + }{ + {"above cap is clamped", 5000, configtables.MaxAuthCodeTTL}, + {"zero falls back to default", 0, configtables.DefaultAuthCodeTTL}, + {"in-range used verbatim", 120, 120}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + h, store, cfg := newIssuanceHandler(t) + cfg.ClientConfig.OAuth2ServerConfig.AuthCodeTTL = tc.configured + cid := seedClient(t, store, []string{"http://127.0.0.1:1234/cb"}) + + v := url.Values{} + v.Set("client_id", cid) + v.Set("redirect_uri", "http://127.0.0.1:1234/cb") + v.Set("response_type", "code") + v.Set("code_challenge", "challenge") + v.Set("code_challenge_method", "S256") + v.Set("resource", testMCPResource) + v.Set("state", "xyz") + + before := time.Now() + ctx := getCtx("/oauth2/authorize?" + v.Encode()) + h.handleAuthorize(ctx) + require.Equal(t, fasthttp.StatusFound, ctx.Response.StatusCode()) + + loc := string(ctx.Response.Header.Peek("Location")) + u, err := url.Parse(loc) + require.NoError(t, err) + flowID := u.Query().Get("flow") + require.NotEmpty(t, flowID) + + req, err := store.GetOAuth2AuthorizeRequestByID(context.Background(), flowID) + require.NoError(t, err) + + wantDeadline := before.Add(time.Duration(tc.wantTTL) * time.Second) + assert.WithinDuration(t, wantDeadline, req.ExpiresAt, 30*time.Second) + }) + } +} diff --git a/transports/bifrost-http/handlers/mcpoauth2jwt.go b/transports/bifrost-http/handlers/mcpoauth2jwt.go new file mode 100644 index 00000000000..7e1852a1ee8 --- /dev/null +++ b/transports/bifrost-http/handlers/mcpoauth2jwt.go @@ -0,0 +1,161 @@ +package handlers + +import ( + "crypto/rsa" + "fmt" + "slices" + "strings" + "sync" + + "github.com/golang-jwt/jwt/v5" + "github.com/maximhq/bifrost/core/schemas" + configtables "github.com/maximhq/bifrost/framework/configstore/tables" + "github.com/maximhq/bifrost/transports/bifrost-http/lib" + "github.com/valyala/fasthttp" +) + +// mcpJWTPublicKeys caches the parsed RSA public key so verification can skip +// re-parsing the PEM on every request. Keyed by the public-key PEM (its content) +// rather than the kid: the content uniquely identifies the keypair, so a rotated +// key materializes a fresh entry while a reused kid across distinct keys can +// never alias the wrong key. The signing key is immutable for the process +// lifetime, so an entry never goes stale. +var mcpJWTPublicKeys sync.Map // publicKeyPEM (string) -> *rsa.PublicKey + +// mcpJWTPublicKey returns the verification public key for the given signing key, +// parsing and caching it on first use. +func mcpJWTPublicKey(signingKey *configtables.OAuth2SigningKey) (*rsa.PublicKey, error) { + if cached, ok := mcpJWTPublicKeys.Load(signingKey.PublicKeyPEM); ok { + return cached.(*rsa.PublicKey), nil + } + pubKey, err := parseRSAPublicKeyPEM(signingKey.PublicKeyPEM) + if err != nil { + return nil, fmt.Errorf("invalid signing key: %w", err) + } + mcpJWTPublicKeys.Store(signingKey.PublicKeyPEM, pubKey) + return pubKey, nil +} + +// jwtMCPClaims are the custom claims embedded in Bifrost-issued /mcp JWTs. +type jwtMCPClaims struct { + jwt.RegisteredClaims + BfMode string `json:"bf_mode"` // user | vk | session + Scope string `json:"scope"` +} + +// extractBearerJWT returns the raw JWT string from an Authorization: Bearer +// header when the token looks like a JWT (starts with "eyJ"). Returns empty +// string when the header is absent or the token is a VK (starts with "sk-bf-"). +func extractBearerJWT(ctx *fasthttp.RequestCtx) string { + auth := strings.TrimSpace(string(ctx.Request.Header.Peek("Authorization"))) + if auth == "" { + return "" + } + if !strings.HasPrefix(strings.ToLower(auth), "bearer ") { + return "" + } + token := strings.TrimSpace(auth[7:]) + // JWTs are base64url-encoded JSON starting with '{', which encodes to "eyJ". + // VKs start with the "sk-bf-" prefix. Anything not starting with "eyJ" is + // treated as a non-JWT credential and left to the VK path. + if !strings.HasPrefix(token, "eyJ") { + return "" + } + return token +} + +// verifyMCPJWT parses and verifies a Bifrost-issued JWT for the /mcp endpoint. +// It validates the RS256 signature using the supplied signing key, checks the +// audience matches the canonical /mcp resource URL (RFC 8707), and returns +// the verified claims. The caller provides the signing key (typically from a +// process-lifetime cache) so verification need not read it per request. +func verifyMCPJWT(ctx *fasthttp.RequestCtx, rawToken string, store *lib.Config, signingKey *configtables.OAuth2SigningKey) (*jwtMCPClaims, error) { + if signingKey == nil { + return nil, fmt.Errorf("signing key unavailable") + } + + pubKey, err := mcpJWTPublicKey(signingKey) + if err != nil { + return nil, err + } + + // Pin the issuer to this instance: the kid + signature checks only prove the + // token was signed by our key, so a different authorization server sharing + // the same keypair would otherwise pass. Issuance stamps iss from the same + // oauth2IssuerURL, so the two always agree. + issuer := oauth2IssuerURL(ctx, store) + + claims := &jwtMCPClaims{} + tok, err := jwt.ParseWithClaims(rawToken, claims, func(t *jwt.Token) (any, error) { + if _, ok := t.Method.(*jwt.SigningMethodRSA); !ok { + return nil, fmt.Errorf("unexpected signing method: %v", t.Header["alg"]) + } + if kid, _ := t.Header["kid"].(string); kid != signingKey.KID { + return nil, fmt.Errorf("unknown key id %q", kid) + } + return pubKey, nil + }, jwt.WithExpirationRequired(), jwt.WithIssuedAt(), jwt.WithIssuer(issuer), + // Accept exactly the algorithm we issue. The SigningMethodRSA type + // assertion above admits the whole RS family (RS256/384/512); pin to + // RS256 so verification matches issuance. + jwt.WithValidMethods([]string{jwt.SigningMethodRS256.Alg()})) + if err != nil { + return nil, fmt.Errorf("invalid token: %w", err) + } + if !tok.Valid { + return nil, fmt.Errorf("token is not valid") + } + // WithIssuedAt only validates iat when present; require it, since every + // token we issue stamps one. + if claims.IssuedAt == nil { + return nil, fmt.Errorf("token missing iat claim") + } + + // RFC 8707: the token must have been issued for this specific resource. + resource := oauth2MCPResourceURL(ctx, store) + aud, err := claims.GetAudience() + if err != nil || !slices.Contains(aud, resource) { + return nil, fmt.Errorf("token audience does not match this resource") + } + + return claims, nil +} + +// injectJWTContext sets the identity context keys from verified JWT claims, +// mirroring what header auth sets today so everything downstream (governance, +// per-user upstream OAuth, tool-group filtering) works unchanged. +// +// bf_mode=user → BifrostContextKeyUserID +// bf_mode=vk → BifrostContextKeyVirtualKey (governance derives the VK row ID from it) +// bf_mode=session → BifrostContextKeyMCPSessionID +func injectJWTContext(bifrostCtx *schemas.BifrostContext, claims *jwtMCPClaims, vk *configtables.TableVirtualKey) error { + sub := claims.Subject + if sub == "" { + return fmt.Errorf("JWT missing sub claim") + } + switch schemas.MCPAuthMode(claims.BfMode) { + case schemas.MCPAuthModeUser: + bifrostCtx.SetValue(schemas.BifrostContextKeyUserID, sub) + case schemas.MCPAuthModeVK: + if vk == nil { + return fmt.Errorf("VK not provided for vk-mode JWT injection") + } + // Set the VK value only. Governance's PreMCPConnectionHook resolves it to + // the VK row ID (BifrostContextKeyGovernanceVirtualKeyID) on the connect + // path before the per-user credential resolver needs it — the same way the + // x-bf-vk header path does, which never stamps the row ID at ingress either. + bifrostCtx.SetValue(schemas.BifrostContextKeyVirtualKey, vk.Value.GetValue()) + case schemas.MCPAuthModeSession: + bifrostCtx.SetValue(schemas.BifrostContextKeyMCPSessionID, sub) + default: + return fmt.Errorf("unknown bf_mode %q in JWT", claims.BfMode) + } + return nil +} + +// wwwAuthenticateValue returns the WWW-Authenticate header value pointing at +// the /mcp resource metadata endpoint, per RFC 9728 §5. +func wwwAuthenticateValue(ctx *fasthttp.RequestCtx, store *lib.Config) string { + base := oauth2IssuerURL(ctx, store) + return fmt.Sprintf(`Bearer resource_metadata="%s/.well-known/oauth-protected-resource/mcp"`, base) +} diff --git a/transports/bifrost-http/handlers/mcpoauth2jwt_test.go b/transports/bifrost-http/handlers/mcpoauth2jwt_test.go new file mode 100644 index 00000000000..2c6e33d64df --- /dev/null +++ b/transports/bifrost-http/handlers/mcpoauth2jwt_test.go @@ -0,0 +1,457 @@ +package handlers + +import ( + "context" + "crypto/rand" + "crypto/rsa" + "crypto/x509" + "encoding/base64" + "encoding/pem" + "strings" + "testing" + "time" + + "github.com/golang-jwt/jwt/v5" + "github.com/maximhq/bifrost/core/schemas" + "github.com/maximhq/bifrost/framework/configstore" + configtables "github.com/maximhq/bifrost/framework/configstore/tables" + "github.com/maximhq/bifrost/transports/bifrost-http/lib" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "github.com/valyala/fasthttp" +) + +// testIssuer is a stable issuer URL so token verification does not depend on the +// request Host header. Configuring it on OAuth2ServerConfig makes oauth2IssuerURL +// deterministic across mint and verify. +const testIssuer = "https://bifrost.test" + +// testMCPResource is the canonical /mcp resource audience derived from the issuer. +const testMCPResource = testIssuer + "/mcp" + +// mockOAuth2Store is an embedded-interface ConfigStore stub for the /mcp auth +// tests. It serves a fixed signing key and a small set of virtual keys; every +// other ConfigStore method panics if a test reaches it unexpectedly. +type mockOAuth2Store struct { + configstore.ConfigStore + signingKey *configtables.OAuth2SigningKey + signingErr error + vksByID map[string]*configtables.TableVirtualKey + vksByValue map[string]*configtables.TableVirtualKey + authReqs map[string]*configtables.TableOAuth2AuthorizeRequest + clients map[string]*configtables.TableOAuth2Client + + sessionRows []configstore.OAuth2SessionRow + sessionByID map[string]*configtables.TableOAuth2RefreshToken + listErr error + revokeErr error + revokedIDs []string +} + +func (m *mockOAuth2Store) GetOAuth2AuthorizeRequestByID(_ context.Context, id string) (*configtables.TableOAuth2AuthorizeRequest, error) { + if r, ok := m.authReqs[id]; ok { + // Return a copy so a handler mutating the loaded struct does not mutate the + // stored row in place — the real store loads a fresh copy from the DB. + cp := *r + return &cp, nil + } + return nil, configstore.ErrNotFound +} + +func (m *mockOAuth2Store) GetOAuth2ClientByClientID(_ context.Context, clientID string) (*configtables.TableOAuth2Client, error) { + if c, ok := m.clients[clientID]; ok { + return c, nil + } + return nil, configstore.ErrNotFound +} + +func (m *mockOAuth2Store) ListOAuth2Sessions(_ context.Context, _ configstore.OAuth2SessionsQueryParams) ([]configstore.OAuth2SessionRow, int64, error) { + if m.listErr != nil { + return nil, 0, m.listErr + } + return m.sessionRows, int64(len(m.sessionRows)), nil +} + +func (m *mockOAuth2Store) GetOAuth2SessionByID(_ context.Context, id string) (*configtables.TableOAuth2RefreshToken, error) { + if r, ok := m.sessionByID[id]; ok { + return r, nil + } + return nil, configstore.ErrNotFound +} + +func (m *mockOAuth2Store) RevokeOAuth2Session(_ context.Context, id string) error { + if m.revokeErr != nil { + return m.revokeErr + } + if _, ok := m.sessionByID[id]; !ok { + return configstore.ErrNotFound + } + m.revokedIDs = append(m.revokedIDs, id) + return nil +} + +func (m *mockOAuth2Store) ConsentOAuth2AuthorizeRequest(_ context.Context, req *configtables.TableOAuth2AuthorizeRequest) error { + existing, ok := m.authReqs[req.ID] + if !ok || existing.Status != configtables.OAuth2AuthorizeRequestStatusPending { + return configstore.ErrNotFound + } + existing.Status = configtables.OAuth2AuthorizeRequestStatusConsented + existing.CodeHash = req.CodeHash + existing.BfMode = req.BfMode + existing.BfSub = req.BfSub + return nil +} + +func (m *mockOAuth2Store) GetOAuth2SigningKey(_ context.Context) (*configtables.OAuth2SigningKey, error) { + if m.signingErr != nil { + return nil, m.signingErr + } + return m.signingKey, nil +} + +func (m *mockOAuth2Store) GetVirtualKey(_ context.Context, id string) (*configtables.TableVirtualKey, error) { + if vk, ok := m.vksByID[id]; ok { + return vk, nil + } + return nil, configstore.ErrNotFound +} + +func (m *mockOAuth2Store) GetVirtualKeyByValue(_ context.Context, value string) (*configtables.TableVirtualKey, error) { + if vk, ok := m.vksByValue[value]; ok { + return vk, nil + } + return nil, configstore.ErrNotFound +} + +// newTestSigningKey generates an RS2048 keypair and returns it both as the stored +// OAuth2SigningKey (PKCS8 PEM) and the raw private key for signing test tokens. +func newTestSigningKey(t *testing.T) (*configtables.OAuth2SigningKey, *rsa.PrivateKey) { + t.Helper() + priv, err := rsa.GenerateKey(rand.Reader, 2048) + require.NoError(t, err) + der, err := x509.MarshalPKCS8PrivateKey(priv) + require.NoError(t, err) + privPEM := string(pem.EncodeToMemory(&pem.Block{Type: "PRIVATE KEY", Bytes: der})) + pubDER, err := x509.MarshalPKIXPublicKey(&priv.PublicKey) + require.NoError(t, err) + pubPEM := string(pem.EncodeToMemory(&pem.Block{Type: "PUBLIC KEY", Bytes: pubDER})) + return &configtables.OAuth2SigningKey{KID: "test-kid", PrivateKeyPEM: privPEM, PublicKeyPEM: pubPEM}, priv +} + +// newTestOAuth2Config builds a lib.Config wired to the given store with a stable +// issuer, for the requested /mcp auth mode and auth-enforcement setting. +func newTestOAuth2Config(store configstore.ConfigStore, authMode configtables.MCPServerAuthMode, enforceAuth bool) *lib.Config { + return &lib.Config{ + ConfigStore: store, + ClientConfig: &configstore.ClientConfig{ + MCPServerAuthMode: authMode, + EnforceAuthOnInference: enforceAuth, + OAuth2ServerConfig: &configtables.OAuth2ServerConfig{ + IssuerURL: schemas.NewSecretVar(testIssuer), + AuthCodeTTL: configtables.DefaultAuthCodeTTL, + AccessTokenTTL: configtables.DefaultAccessTokenTTL, + }, + }, + } +} + +// mintTestToken signs a token with valid defaults (vk mode), applying mutate to +// override claims/header for negative cases. signMethod and signKey let callers +// force a non-RS256 algorithm or a wrong signing key. +func mintTestToken(t *testing.T, priv *rsa.PrivateKey, kid string, mutate func(jwt.MapClaims)) string { + t.Helper() + now := time.Now() + claims := jwt.MapClaims{ + "iss": testIssuer, + "aud": jwt.ClaimStrings{testMCPResource}, + "sub": "vk-123", + "bf_mode": string(schemas.MCPAuthModeVK), + "scope": "mcp", + "iat": now.Unix(), + "nbf": now.Unix(), + "exp": now.Add(10 * time.Minute).Unix(), + } + if mutate != nil { + mutate(claims) + } + tok := jwt.NewWithClaims(jwt.SigningMethodRS256, claims) + tok.Header["kid"] = kid + signed, err := tok.SignedString(priv) + require.NoError(t, err) + return signed +} + +func TestExtractBearerJWT(t *testing.T) { + cases := []struct { + name string + header string + want string + }{ + {name: "jwt bearer", header: "Bearer eyJhbGciOiJSUzI1NiJ9.x.y", want: "eyJhbGciOiJSUzI1NiJ9.x.y"}, + {name: "case-insensitive scheme", header: "bearer eyJabc", want: "eyJabc"}, + {name: "virtual key is not a jwt", header: "Bearer sk-bf-abc123", want: ""}, + {name: "non-bearer scheme", header: "Basic eyJabc", want: ""}, + {name: "empty header", header: "", want: ""}, + {name: "bearer without token", header: "Bearer ", want: ""}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + ctx := &fasthttp.RequestCtx{} + if tc.header != "" { + ctx.Request.Header.Set("Authorization", tc.header) + } + assert.Equal(t, tc.want, extractBearerJWT(ctx)) + }) + } +} + +func TestVerifyMCPJWT_ValidEachMode(t *testing.T) { + key, priv := newTestSigningKey(t) + store := &mockOAuth2Store{signingKey: key} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + + modes := []struct { + mode schemas.MCPAuthMode + sub string + }{ + {schemas.MCPAuthModeUser, "user-1"}, + {schemas.MCPAuthModeVK, "vk-123"}, + {schemas.MCPAuthModeSession, "session-abc"}, + } + for _, m := range modes { + t.Run(string(m.mode), func(t *testing.T) { + raw := mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["bf_mode"] = string(m.mode) + c["sub"] = m.sub + }) + ctx := &fasthttp.RequestCtx{} + claims, err := verifyMCPJWT(ctx, raw, cfg, key) + require.NoError(t, err) + require.NotNil(t, claims) + assert.Equal(t, string(m.mode), claims.BfMode) + assert.Equal(t, m.sub, claims.Subject) + }) + } +} + +func TestVerifyMCPJWT_Rejections(t *testing.T) { + key, priv := newTestSigningKey(t) + _, otherPriv := newTestSigningKey(t) + store := &mockOAuth2Store{signingKey: key} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + + cases := []struct { + name string + // raw builds the token string under test. + raw func() string + }{ + { + name: "expired token", + raw: func() string { + return mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["exp"] = time.Now().Add(-time.Minute).Unix() + }) + }, + }, + { + name: "nbf in the future", + raw: func() string { + return mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["nbf"] = time.Now().Add(time.Hour).Unix() + }) + }, + }, + { + name: "missing exp", + raw: func() string { + return mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + delete(c, "exp") + }) + }, + }, + { + name: "missing iat", + raw: func() string { + return mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + delete(c, "iat") + }) + }, + }, + { + name: "issuer mismatch", + raw: func() string { + return mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["iss"] = "https://evil.example" + }) + }, + }, + { + name: "audience mismatch", + raw: func() string { + return mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["aud"] = jwt.ClaimStrings{"https://bifrost.test/other"} + }) + }, + }, + { + name: "unknown kid", + raw: func() string { + return mintTestToken(t, priv, "wrong-kid", nil) + }, + }, + { + name: "wrong signing key", + raw: func() string { + return mintTestToken(t, otherPriv, key.KID, nil) + }, + }, + { + name: "non-RS256 algorithm (HS256)", + raw: func() string { + tok := jwt.NewWithClaims(jwt.SigningMethodHS256, jwt.MapClaims{ + "iss": testIssuer, "aud": jwt.ClaimStrings{testMCPResource}, + "sub": "vk-123", "bf_mode": "vk", "iat": time.Now().Unix(), + "nbf": time.Now().Unix(), "exp": time.Now().Add(time.Hour).Unix(), + }) + tok.Header["kid"] = key.KID + signed, err := tok.SignedString([]byte("hs-secret")) + require.NoError(t, err) + return signed + }, + }, + { + name: "RS-family but not RS256 (RS384)", + raw: func() string { + tok := jwt.NewWithClaims(jwt.SigningMethodRS384, jwt.MapClaims{ + "iss": testIssuer, "aud": jwt.ClaimStrings{testMCPResource}, + "sub": "vk-123", "bf_mode": "vk", "iat": time.Now().Unix(), + "nbf": time.Now().Unix(), "exp": time.Now().Add(time.Hour).Unix(), + }) + tok.Header["kid"] = key.KID + signed, err := tok.SignedString(priv) + require.NoError(t, err) + return signed + }, + }, + { + name: "alg none", + raw: func() string { + header := base64.RawURLEncoding.EncodeToString([]byte(`{"alg":"none","typ":"JWT","kid":"test-kid"}`)) + payload := base64.RawURLEncoding.EncodeToString([]byte(`{"iss":"https://bifrost.test","aud":["https://bifrost.test/mcp"],"sub":"vk-123","bf_mode":"vk","iat":1700000000,"nbf":1700000000,"exp":9999999999}`)) + return header + "." + payload + "." + }, + }, + { + name: "malformed garbage", + raw: func() string { + return "eyJ.not-a-valid.token" + }, + }, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + ctx := &fasthttp.RequestCtx{} + claims, err := verifyMCPJWT(ctx, tc.raw(), cfg, key) + require.Error(t, err) + assert.Nil(t, claims) + }) + } +} + +// TestVerifyMCPJWT_NilSigningKeyNotLabeledInvalidToken pins that a nil signing +// key — the caller's signal that loading the key failed (no config store, +// missing key) — surfaces as a config fault, never as the caller's token being +// invalid. +func TestVerifyMCPJWT_NilSigningKeyNotLabeledInvalidToken(t *testing.T) { + key, priv := newTestSigningKey(t) + raw := mintTestToken(t, priv, key.KID, nil) + cfg := newTestOAuth2Config(&mockOAuth2Store{signingKey: key}, configtables.MCPServerAuthModeBoth, false) + + ctx := &fasthttp.RequestCtx{} + _, err := verifyMCPJWT(ctx, raw, cfg, nil) + require.Error(t, err) + assert.Contains(t, err.Error(), "signing key unavailable") + assert.NotContains(t, err.Error(), "invalid token") +} + +// TestCachedSigningKey_ConfigFaults pins that the handler's key loader — which +// now owns reading the signing key for the JWT verify path — surfaces +// infrastructure faults distinctly, never as the caller's token being invalid. +func TestCachedSigningKey_ConfigFaults(t *testing.T) { + t.Run("nil config store", func(t *testing.T) { + cfg := newTestOAuth2Config(nil, configtables.MCPServerAuthModeBoth, false) + cfg.ConfigStore = nil + h := &MCPServerHandler{config: cfg} + _, err := h.config.GetOAuth2SigningKey(bgCtx()) + require.Error(t, err) + assert.Equal(t, "config store unavailable", err.Error()) + assert.NotContains(t, err.Error(), "invalid token") + }) + + t.Run("signing key load error", func(t *testing.T) { + store := &mockOAuth2Store{signingErr: configstore.ErrNotFound} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + h := &MCPServerHandler{config: cfg} + _, err := h.config.GetOAuth2SigningKey(bgCtx()) + require.Error(t, err) + assert.NotContains(t, err.Error(), "invalid token") + }) +} + +func TestInjectJWTContext(t *testing.T) { + activeVK := &configtables.TableVirtualKey{ID: "vk-row-1", Value: *schemas.NewSecretVar("sk-bf-active")} + + t.Run("user mode sets user id", func(t *testing.T) { + bc := schemas.NewBifrostContext(context.Background(), time.Time{}) + err := injectJWTContext(bc, &jwtMCPClaims{ + RegisteredClaims: jwt.RegisteredClaims{Subject: "user-1"}, BfMode: "user", + }, nil) + require.NoError(t, err) + assert.Equal(t, "user-1", bc.Value(schemas.BifrostContextKeyUserID)) + }) + + t.Run("vk mode sets the raw vk value and lets governance derive the id", func(t *testing.T) { + bc := schemas.NewBifrostContext(context.Background(), time.Time{}) + err := injectJWTContext(bc, &jwtMCPClaims{ + RegisteredClaims: jwt.RegisteredClaims{Subject: "vk-row-1"}, BfMode: "vk", + }, activeVK) + require.NoError(t, err) + assert.Equal(t, "sk-bf-active", bc.Value(schemas.BifrostContextKeyVirtualKey)) + // The VK row ID is resolved later by governance's PreMCPConnectionHook from + // the value, not stamped here — mirrors the x-bf-vk header path. + assert.Nil(t, bc.Value(schemas.BifrostContextKeyGovernanceVirtualKeyID)) + }) + + t.Run("vk mode without vk errors", func(t *testing.T) { + bc := schemas.NewBifrostContext(context.Background(), time.Time{}) + err := injectJWTContext(bc, &jwtMCPClaims{ + RegisteredClaims: jwt.RegisteredClaims{Subject: "vk-row-1"}, BfMode: "vk", + }, nil) + require.Error(t, err) + }) + + t.Run("session mode sets session id", func(t *testing.T) { + bc := schemas.NewBifrostContext(context.Background(), time.Time{}) + err := injectJWTContext(bc, &jwtMCPClaims{ + RegisteredClaims: jwt.RegisteredClaims{Subject: "session-abc"}, BfMode: "session", + }, nil) + require.NoError(t, err) + assert.Equal(t, "session-abc", bc.Value(schemas.BifrostContextKeyMCPSessionID)) + }) + + t.Run("missing sub errors", func(t *testing.T) { + bc := schemas.NewBifrostContext(context.Background(), time.Time{}) + err := injectJWTContext(bc, &jwtMCPClaims{BfMode: "user"}, nil) + require.Error(t, err) + }) + + t.Run("unknown bf_mode errors", func(t *testing.T) { + bc := schemas.NewBifrostContext(context.Background(), time.Time{}) + err := injectJWTContext(bc, &jwtMCPClaims{ + RegisteredClaims: jwt.RegisteredClaims{Subject: "x"}, BfMode: "bogus", + }, nil) + require.Error(t, err) + assert.True(t, strings.Contains(err.Error(), "bf_mode")) + }) +} diff --git a/transports/bifrost-http/handlers/mcpoauth2sessions.go b/transports/bifrost-http/handlers/mcpoauth2sessions.go new file mode 100644 index 00000000000..9892e2b342e --- /dev/null +++ b/transports/bifrost-http/handlers/mcpoauth2sessions.go @@ -0,0 +1,185 @@ +package handlers + +import ( + "errors" + "strconv" + "strings" + + "github.com/fasthttp/router" + "github.com/maximhq/bifrost/core/schemas" + "github.com/maximhq/bifrost/framework/configstore" + "github.com/maximhq/bifrost/transports/bifrost-http/lib" + "github.com/valyala/fasthttp" +) + +// OAuth2SessionsHandler serves the Connected Clients API used by the sessions +// management UI to list active downstream grants and revoke them. +type OAuth2SessionsHandler struct { + store *lib.Config +} + +// NewOAuth2SessionsHandler creates a new sessions handler. +func NewOAuth2SessionsHandler(store *lib.Config) *OAuth2SessionsHandler { + return &OAuth2SessionsHandler{store: store} +} + +// RegisterRoutes wires the Connected Clients endpoints. +func (h *OAuth2SessionsHandler) RegisterRoutes(r *router.Router, middlewares ...schemas.BifrostHTTPMiddleware) { + r.GET("/api/oauth2/sessions", lib.ChainMiddlewares(h.listSessions, middlewares...)) + r.DELETE("/api/oauth2/sessions/{id}", lib.ChainMiddlewares(h.revokeSession, middlewares...)) +} + +const ( + oauth2SessionsDefaultLimit = 50 + oauth2SessionsMaxLimit = 500 +) + +// oauth2SessionsListResponse is the wire shape for GET /api/oauth2/sessions. +// Mirrors the MCP auth-sessions list contract (sessions + count/total_count/ +// limit/offset) so the grants UI paginates the same way. +type oauth2SessionsListResponse struct { + Sessions []configstore.OAuth2SessionRow `json:"sessions"` + Count int `json:"count"` + TotalCount int `json:"total_count"` + Limit int `json:"limit"` + Offset int `json:"offset"` +} + +// oauth2SessionsListQuery is the parsed query string for GET /api/oauth2/sessions. +// Modes filters on bf_mode (user/vk/session); Search is a case-insensitive +// substring matched against the client name/id and the bound identity. +// Limit/Offset paginate the filtered result. +type oauth2SessionsListQuery struct { + Search string + Modes []string + Limit int + Offset int +} + +// parseOAuth2SessionsListQuery extracts pagination + filter params from the +// request query string. On validation failure it writes a 400 response and +// returns ok=false — the caller must early-return without further writes. +func parseOAuth2SessionsListQuery(ctx *fasthttp.RequestCtx) (oauth2SessionsListQuery, bool) { + q := oauth2SessionsListQuery{Limit: oauth2SessionsDefaultLimit} + args := ctx.QueryArgs() + q.Search = strings.TrimSpace(string(args.Peek("q"))) + q.Modes = parseCommaSeparated(string(args.Peek("bf_mode"))) + // bf_mode only ever holds user/vk/session. Reject anything else with a 400 + // rather than letting an unknown mode silently match no rows (the SQL filter + // would just return an empty page, hiding the typo from the caller). + for _, mode := range q.Modes { + switch mode { + case "user", "vk", "session": + default: + SendError(ctx, fasthttp.StatusBadRequest, "Invalid bf_mode parameter: must be one or more of user, vk, session") + return q, false + } + } + if s := string(args.Peek("limit")); s != "" { + n, err := strconv.Atoi(s) + if err != nil { + SendError(ctx, fasthttp.StatusBadRequest, "Invalid limit parameter: must be a number") + return q, false + } + if n <= 0 { + SendError(ctx, fasthttp.StatusBadRequest, "Invalid limit parameter: must be greater than zero") + return q, false + } + if n > oauth2SessionsMaxLimit { + n = oauth2SessionsMaxLimit + } + q.Limit = n + } + if s := string(args.Peek("offset")); s != "" { + n, err := strconv.Atoi(s) + if err != nil { + SendError(ctx, fasthttp.StatusBadRequest, "Invalid offset parameter: must be a number") + return q, false + } + if n < 0 { + SendError(ctx, fasthttp.StatusBadRequest, "Invalid offset parameter: must be non-negative") + return q, false + } + q.Offset = n + } + return q, true +} + +// GET /api/oauth2/sessions — list active downstream grants, filtered + paginated. +// Filtering (search + mode) and pagination (limit/offset) are pushed to SQL by +// the store; the single-table source has no cross-table merge, so — unlike the +// MCP auth-sessions handler — there is nothing to slice here. The store also +// returns the total count matching the filters, used for the page indicator. +func (h *OAuth2SessionsHandler) listSessions(ctx *fasthttp.RequestCtx) { + if h.store.ConfigStore == nil { + SendError(ctx, fasthttp.StatusServiceUnavailable, "config store unavailable") + return + } + q, ok := parseOAuth2SessionsListQuery(ctx) + if !ok { + return + } + sessions, totalCount, err := h.store.ConfigStore.ListOAuth2Sessions(ctx, configstore.OAuth2SessionsQueryParams{ + Search: q.Search, + Modes: q.Modes, + Limit: q.Limit, + Offset: q.Offset, + }) + if err != nil { + logger.Error("oauth2 sessions: failed to list sessions: %v", err) + SendError(ctx, fasthttp.StatusInternalServerError, "failed to list sessions") + return + } + + SendJSON(ctx, oauth2SessionsListResponse{ + Sessions: sessions, + Count: len(sessions), + TotalCount: int(totalCount), + Limit: q.Limit, + Offset: q.Offset, + }) +} + +// DELETE /api/oauth2/sessions/{id} — revoke a specific downstream grant. +func (h *OAuth2SessionsHandler) revokeSession(ctx *fasthttp.RequestCtx) { + if h.store.ConfigStore == nil { + SendError(ctx, fasthttp.StatusServiceUnavailable, "config store unavailable") + return + } + id, ok := ctx.UserValue("id").(string) + if !ok || id == "" { + SendError(ctx, fasthttp.StatusBadRequest, "invalid session id") + return + } + + // Authorization is by visibility: the scoped read returns not-found for rows + // outside the caller's scope, and RevokeOAuth2Session itself is unscoped, so + // this load is what stops a caller from revoking a grant they cannot see. + // + // Revoke is intentionally not restricted beyond that. It's destructive cleanup + // — it only stamps revoked_at and never acts under the grant's identity — so + // any caller who can see a row (its owner, a team lead, or an admin) may revoke + // it, across all modes. This matches the per-user MCP session revoke. + // Re-authentication is gated separately to the bound user, because that path + // mints credentials under the user's identity; revoke does not. + if _, err := h.store.ConfigStore.GetOAuth2SessionByID(ctx, id); err != nil { + if errors.Is(err, configstore.ErrNotFound) { + SendError(ctx, fasthttp.StatusNotFound, "session not found or already revoked") + return + } + logger.Error("oauth2 sessions: failed to load session: %v", err) + SendError(ctx, fasthttp.StatusInternalServerError, "failed to load session") + return + } + + if err := h.store.ConfigStore.RevokeOAuth2Session(ctx, id); err != nil { + if errors.Is(err, configstore.ErrNotFound) { + SendError(ctx, fasthttp.StatusNotFound, "session not found or already revoked") + return + } + logger.Error("oauth2 sessions: failed to revoke session: %v", err) + SendError(ctx, fasthttp.StatusInternalServerError, "failed to revoke session") + return + } + ctx.SetStatusCode(fasthttp.StatusNoContent) +} diff --git a/transports/bifrost-http/handlers/mcpoauth2sessions_test.go b/transports/bifrost-http/handlers/mcpoauth2sessions_test.go new file mode 100644 index 00000000000..25a38ee4ba2 --- /dev/null +++ b/transports/bifrost-http/handlers/mcpoauth2sessions_test.go @@ -0,0 +1,109 @@ +package handlers + +import ( + "encoding/json" + "errors" + "testing" + + "github.com/maximhq/bifrost/core/schemas" + "github.com/maximhq/bifrost/framework/configstore" + configtables "github.com/maximhq/bifrost/framework/configstore/tables" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "github.com/valyala/fasthttp" +) + +func newSessionsHandler(store *mockOAuth2Store) *OAuth2SessionsHandler { + SetLogger(&mockLogger{}) + return NewOAuth2SessionsHandler(newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false)) +} + +func TestListSessions(t *testing.T) { + t.Run("returns the grant rows", func(t *testing.T) { + store := &mockOAuth2Store{sessionRows: []configstore.OAuth2SessionRow{ + {ID: "s1", ClientID: "c1", BfMode: "vk", BfSub: "vk-1", BfSubDisplay: "Alpha VK"}, + }} + h := newSessionsHandler(store) + ctx := &fasthttp.RequestCtx{} + h.listSessions(ctx) + require.Equal(t, fasthttp.StatusOK, ctx.Response.StatusCode()) + + var resp struct { + Sessions []configstore.OAuth2SessionRow `json:"sessions"` + } + require.NoError(t, json.Unmarshal(ctx.Response.Body(), &resp)) + require.Len(t, resp.Sessions, 1) + assert.Equal(t, "Alpha VK", resp.Sessions[0].BfSubDisplay) + }) + + t.Run("store error surfaces 500", func(t *testing.T) { + h := newSessionsHandler(&mockOAuth2Store{listErr: errors.New("boom")}) + ctx := &fasthttp.RequestCtx{} + h.listSessions(ctx) + assert.Equal(t, fasthttp.StatusInternalServerError, ctx.Response.StatusCode()) + }) +} + +func TestRevokeSession(t *testing.T) { + newStore := func(row *configtables.TableOAuth2RefreshToken) *mockOAuth2Store { + return &mockOAuth2Store{sessionByID: map[string]*configtables.TableOAuth2RefreshToken{row.ID: row}} + } + revokeCtx := func(id, callerUserID string) *fasthttp.RequestCtx { + ctx := &fasthttp.RequestCtx{} + ctx.SetUserValue("id", id) + if callerUserID != "" { + ctx.SetUserValue(schemas.BifrostContextKeyUserID, callerUserID) + } + return ctx + } + + t.Run("vk-mode grant revokes without an identity gate", func(t *testing.T) { + store := newStore(&configtables.TableOAuth2RefreshToken{ID: "s1", BfMode: "vk", BfSub: "vk-1"}) + h := newSessionsHandler(store) + ctx := revokeCtx("s1", "") + h.revokeSession(ctx) + assert.Equal(t, fasthttp.StatusNoContent, ctx.Response.StatusCode()) + assert.Contains(t, store.revokedIDs, "s1") + }) + + t.Run("user-mode grant revokes when the caller matches bf_sub", func(t *testing.T) { + store := newStore(&configtables.TableOAuth2RefreshToken{ID: "s1", BfMode: "user", BfSub: "user-1"}) + h := newSessionsHandler(store) + ctx := revokeCtx("s1", "user-1") + h.revokeSession(ctx) + assert.Equal(t, fasthttp.StatusNoContent, ctx.Response.StatusCode()) + }) + + t.Run("revokes a visible grant regardless of caller identity", func(t *testing.T) { + // The handler applies no local identity gate — caller identity is irrelevant + // to revoke, which is destructive cleanup, not an action under the grant's + // identity. Authorization is the scoped store read: a row the caller cannot + // see comes back not-found (covered by the case below), so any caller who can + // load the row may revoke it, across all modes. Here a non-owner succeeds. + store := newStore(&configtables.TableOAuth2RefreshToken{ID: "s1", BfMode: "user", BfSub: "user-1"}) + h := newSessionsHandler(store) + ctx := revokeCtx("s1", "intruder") + h.revokeSession(ctx) + assert.Equal(t, fasthttp.StatusNoContent, ctx.Response.StatusCode()) + assert.Contains(t, store.revokedIDs, "s1") + }) + + t.Run("not-visible grant returns 404 without attempting revoke", func(t *testing.T) { + // A row outside the caller's scope comes back not-found from the scoped read. + // The handler must 404 and must not fall through to revoke — this is where + // visibility is actually enforced. + store := &mockOAuth2Store{sessionByID: map[string]*configtables.TableOAuth2RefreshToken{}} + h := newSessionsHandler(store) + ctx := revokeCtx("missing", "") + h.revokeSession(ctx) + assert.Equal(t, fasthttp.StatusNotFound, ctx.Response.StatusCode()) + assert.Empty(t, store.revokedIDs) + }) + + t.Run("empty id returns 400", func(t *testing.T) { + h := newSessionsHandler(&mockOAuth2Store{}) + ctx := revokeCtx("", "") + h.revokeSession(ctx) + assert.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + }) +} diff --git a/transports/bifrost-http/handlers/mcpoauth2utils.go b/transports/bifrost-http/handlers/mcpoauth2utils.go new file mode 100644 index 00000000000..6a66cb54a5c --- /dev/null +++ b/transports/bifrost-http/handlers/mcpoauth2utils.go @@ -0,0 +1,45 @@ +package handlers + +import ( + "strings" + + configtables "github.com/maximhq/bifrost/framework/configstore/tables" + "github.com/maximhq/bifrost/transports/bifrost-http/lib" + "github.com/valyala/fasthttp" +) + +// oauth2IssuerURL resolves the effective AS issuer URL for a request. +// Uses the explicitly configured IssuerURL when set; falls back to deriving +// it from the request Host header for single-host / dev deployments. +func oauth2IssuerURL(ctx *fasthttp.RequestCtx, store *lib.Config) string { + store.Mu.RLock() + cfg := store.ClientConfig.OAuth2ServerConfig + store.Mu.RUnlock() + if cfg != nil && cfg.IssuerURL.IsSet() { + return cfg.IssuerURL.GetValue() + } + return lib.BuildBaseURL(ctx, "") +} + +// oauth2MCPResourceURL returns the canonical RFC 8707 resource identifier for +// the /mcp endpoint — the single protected resource this server issues tokens +// for. Discovery advertises it, the authorize endpoint pins the request's +// resource parameter to it, and /mcp token verification checks the audience +// against it; routing all three through here keeps them from drifting. +func oauth2MCPResourceURL(ctx *fasthttp.RequestCtx, store *lib.Config) string { + // Trim a trailing slash so a slash-suffixed issuer_url can't produce "//mcp" + // and drift this canonical resource away from what clients normalize to. + return strings.TrimRight(oauth2IssuerURL(ctx, store), "/") + "/mcp" +} + +// oauth2ServerCfg returns the OAuth2 AS-specific config under the read lock, +// falling back to sensible defaults when not yet configured. +func oauth2ServerCfg(store *lib.Config) *configtables.OAuth2ServerConfig { + store.Mu.RLock() + cfg := store.ClientConfig.OAuth2ServerConfig + store.Mu.RUnlock() + if cfg == nil { + return configtables.DefaultOAuth2ServerConfig() + } + return cfg +} diff --git a/transports/bifrost-http/handlers/mcpoauth2utils_test.go b/transports/bifrost-http/handlers/mcpoauth2utils_test.go new file mode 100644 index 00000000000..03c94c197ab --- /dev/null +++ b/transports/bifrost-http/handlers/mcpoauth2utils_test.go @@ -0,0 +1,64 @@ +package handlers + +import ( + "testing" + + "github.com/maximhq/bifrost/framework/configstore" + configtables "github.com/maximhq/bifrost/framework/configstore/tables" + "github.com/maximhq/bifrost/transports/bifrost-http/lib" + "github.com/stretchr/testify/assert" + "github.com/valyala/fasthttp" +) + +func TestMatchRedirectURI(t *testing.T) { + cases := []struct { + name string + candidate string + registered []string + want bool + }{ + {"exact non-loopback match", "https://app.example/cb", []string{"https://app.example/cb"}, true}, + {"non-loopback mismatch", "https://app.example/other", []string{"https://app.example/cb"}, false}, + {"loopback any port (127.0.0.1)", "http://127.0.0.1:55555/cb", []string{"http://127.0.0.1:1234/cb"}, true}, + {"loopback any port (localhost)", "http://localhost:9999/cb", []string{"http://localhost:3000/cb"}, true}, + {"loopback path must still match", "http://127.0.0.1:5/other", []string{"http://127.0.0.1:1/cb"}, false}, + {"loopback scheme must still match", "https://127.0.0.1:5/cb", []string{"http://127.0.0.1:1/cb"}, false}, + {"malformed candidate", "://bad", []string{"https://app.example/cb"}, false}, + {"no registered uris", "https://app.example/cb", nil, false}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + assert.Equal(t, tc.want, matchRedirectURI(tc.candidate, tc.registered)) + }) + } +} + +func TestOAuth2IssuerURL(t *testing.T) { + t.Run("uses the configured issuer when set", func(t *testing.T) { + store := &mockOAuth2Store{} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + assert.Equal(t, testIssuer, oauth2IssuerURL(&fasthttp.RequestCtx{}, cfg)) + }) + + t.Run("falls back to the request host when issuer is unset", func(t *testing.T) { + cfg := &lib.Config{ + ConfigStore: &mockOAuth2Store{}, + ClientConfig: &configstore.ClientConfig{MCPServerAuthMode: configtables.MCPServerAuthModeBoth}, + } + ctx := &fasthttp.RequestCtx{} + ctx.Request.SetRequestURI("http://mcp.local:8080/oauth2/authorize") + ctx.Request.Header.SetHost("mcp.local:8080") + got := oauth2IssuerURL(ctx, cfg) + assert.Equal(t, "http://mcp.local:8080", got) + }) +} + +func TestOAuth2ServerCfg_DefaultsWhenUnset(t *testing.T) { + cfg := &lib.Config{ + ConfigStore: &mockOAuth2Store{}, + ClientConfig: &configstore.ClientConfig{MCPServerAuthMode: configtables.MCPServerAuthModeBoth}, + } + got := oauth2ServerCfg(cfg) + assert.Equal(t, configtables.DefaultAuthCodeTTL, got.AuthCodeTTL) + assert.Equal(t, configtables.DefaultAccessTokenTTL, got.AccessTokenTTL) +} diff --git a/transports/bifrost-http/handlers/mcpserver.go b/transports/bifrost-http/handlers/mcpserver.go index b5bc6120382..f31183367a8 100644 --- a/transports/bifrost-http/handlers/mcpserver.go +++ b/transports/bifrost-http/handlers/mcpserver.go @@ -35,6 +35,14 @@ type MCPToolManager interface { ExecuteResponsesMCPTool(ctx context.Context, toolCall *schemas.ResponsesToolMessage) (*schemas.ResponsesMessage, *schemas.BifrostError) } +// VirtualKeyCache resolves a virtual key by its row ID from an in-memory cache, +// letting the JWT auth path avoid a per-request database read. Satisfied by the +// governance plugin's in-memory store. Optional: when nil (or a cache miss), the +// handler falls back to the config store. +type VirtualKeyCache interface { + GetVirtualKeyByID(ctx context.Context, vkID string) (*tables.TableVirtualKey, bool) +} + // MCPServerHandler manages HTTP requests for MCP server operations // It implements the MCP protocol over HTTP streaming (SSE) for MCP clients type MCPServerHandler struct { @@ -42,11 +50,40 @@ type MCPServerHandler struct { globalMCPServer *server.MCPServer vkMCPServers map[string]*server.MCPServer // Map of vk value -> mcp server config *lib.Config - mu sync.RWMutex + // identityResolver scopes a user-mode /mcp request to the user's own tools by + // resolving a representative virtual key. Optional: when nil, user-mode + // requests fall back to the global server. + identityResolver OAuth2IdentityResolver + // vkCache serves by-ID virtual key lookups on the JWT auth path from the + // governance in-memory store, avoiding a per-request DB read. Optional: a nil + // cache or a miss falls back to the config store. See getVirtualKeyByID. + vkCache VirtualKeyCache + mu sync.RWMutex +} + +// getVirtualKeyByID resolves a virtual key by its row ID for the JWT auth path, +// preferring the governance in-memory cache and falling back to the config store +// on a miss (e.g. a key created since the cache last refreshed) or when no cache +// is wired. The active-state check is left to the caller, matching both sources +// (neither filters inactive keys by ID). +func (h *MCPServerHandler) getVirtualKeyByID(ctx context.Context, vkID string) (*tables.TableVirtualKey, error) { + if h.vkCache != nil { + if vk, ok := h.vkCache.GetVirtualKeyByID(ctx, vkID); ok && vk != nil { + return vk, nil + } + } + if h.config.ConfigStore == nil { + return nil, fmt.Errorf("virtual key not found or inactive") + } + vk, err := h.config.ConfigStore.GetVirtualKey(ctx, vkID) + if err != nil || vk == nil { + return nil, fmt.Errorf("virtual key not found or inactive") + } + return vk, nil } // NewMCPServerHandler creates a new MCP server handler instance -func NewMCPServerHandler(ctx context.Context, config *lib.Config, toolManager MCPToolManager) (*MCPServerHandler, error) { +func NewMCPServerHandler(ctx context.Context, config *lib.Config, toolManager MCPToolManager, identityResolver OAuth2IdentityResolver, vkCache VirtualKeyCache) (*MCPServerHandler, error) { if config == nil { return nil, fmt.Errorf("config is required") } @@ -62,10 +99,12 @@ func NewMCPServerHandler(ctx context.Context, config *lib.Config, toolManager MC ) handler := &MCPServerHandler{ - toolManager: toolManager, - globalMCPServer: globalMCPServer, - config: config, - vkMCPServers: make(map[string]*server.MCPServer), + toolManager: toolManager, + globalMCPServer: globalMCPServer, + config: config, + vkMCPServers: make(map[string]*server.MCPServer), + identityResolver: identityResolver, + vkCache: vkCache, } // Register per-request tool filter so x-bf-mcp-include-clients and x-bf-mcp-include-tools are respected on tools/list @@ -78,36 +117,49 @@ func NewMCPServerHandler(ctx context.Context, config *lib.Config, toolManager MC return nil, fmt.Errorf("failed to sync all MCP servers: %w", err) } + // Warm the signing-key cache when OAuth discovery is enabled: this creates the + // key if absent and populates the cache, so the first JWKS/issuance/verify + // request need not pay the load. This is the single startup warm path for both + // OSS and enterprise. Best-effort — the verify path lazily loads it on a miss — + // but a failure is logged since a persistent one means OAuth cannot work. + if config.ClientConfig.IsMCPOAuthDiscoveryEnabled() { + if _, err := handler.config.GetOAuth2SigningKey(ctx); err != nil { + logger.Warn("mcp: failed to warm oauth2 signing key: %v", err) + } + } + return handler, nil } -// RegisterRoutes registers the MCP server route +// RegisterRoutes registers the MCP server routes. func (h *MCPServerHandler) RegisterRoutes(r *router.Router, middlewares ...schemas.BifrostHTTPMiddleware) { // MCP server endpoint - supports both POST (JSON-RPC) and GET (SSE) r.POST("/mcp", lib.ChainMiddlewares(h.handleMCPServer, middlewares...)) r.GET("/mcp", lib.ChainMiddlewares(h.handleMCPServerSSE, middlewares...)) - // Bifrost is NOT an OAuth authorization server — auth is via the - // x-bf-vk / x-bf-mcp-session-id headers or upstream SSO. Claude Code's - // MCP client may proactively POST `/register` (RFC 7591 DCR) on - // `claude mcp add` and log "SDK auth failed: ..." when the probe fails, - // even though the underlying `/mcp` connection works fine. That warning - // is a known Claude Code bug — see - // https://github.com/anthropics/claude-code/issues/46640 — and is safe - // to ignore. We intentionally do NOT implement an OAuth stub. } func (h *MCPServerHandler) handleMCPServer(ctx *fasthttp.RequestCtx) { - mcpServer, err := h.getMCPServerForRequest(ctx) + authResult, err := h.getMCPServerForRequest(ctx) if err != nil { SendError(ctx, fasthttp.StatusUnauthorized, err.Error()) return } - // Convert context bifrostCtx, cancel := lib.ConvertToBifrostContext(ctx, h.config) bifrostCtx.SetValue(schemas.BifrostContextKeyIsMCPGateway, true) defer cancel() + // Inject JWT identity into BifrostContext so downstream resolvers + // (per-user OAuth, governance, tool-group filtering) see the same context + // keys as header-based auth paths. + if authResult.jwtClaims != nil { + if injErr := injectJWTContext(bifrostCtx, authResult.jwtClaims, authResult.jwtVK); injErr != nil { + SendError(ctx, fasthttp.StatusUnauthorized, injErr.Error()) + return + } + } + mcpServer := authResult.mcpServer + // Use mcp-go server to handle the request // HandleMessage processes JSON-RPC messages and returns appropriate responses response := mcpServer.HandleMessage(bifrostCtx, ctx.PostBody()) @@ -132,12 +184,11 @@ func (h *MCPServerHandler) handleMCPServer(ctx *fasthttp.RequestCtx) { // handleMCPServerSSE handles GET requests for MCP Server-Sent Events streaming func (h *MCPServerHandler) handleMCPServerSSE(ctx *fasthttp.RequestCtx) { - _, err := h.getMCPServerForRequest(ctx) + authResult, err := h.getMCPServerForRequest(ctx) if err != nil { SendError(ctx, fasthttp.StatusUnauthorized, err.Error()) return } - // Signal to transport-plugin and tracing middlewares that this is a streaming // response. Without this, fasthttpResponseToHTTPResponse calls ctx.Response.Body() // during post-hook processing, which materializes the SSE body stream and @@ -166,6 +217,14 @@ func (h *MCPServerHandler) handleMCPServerSSE(ctx *fasthttp.RequestCtx) { bifrostCtx, cancel := lib.ConvertToBifrostContext(ctx, h.config) bifrostCtx.SetValue(schemas.BifrostContextKeyIsMCPGateway, true) + if authResult.jwtClaims != nil { + if injErr := injectJWTContext(bifrostCtx, authResult.jwtClaims, authResult.jwtVK); injErr != nil { + cancel() + SendError(ctx, fasthttp.StatusUnauthorized, injErr.Error()) + return + } + } + // Use SSEStreamReader to bypass fasthttp's internal pipe batching reader := lib.NewSSEStreamReader() ctx.Response.SetBodyStream(reader, -1) @@ -289,7 +348,7 @@ func (h *MCPServerHandler) SyncAllMCPServers(ctx context.Context) error { func (h *MCPServerHandler) SyncVKMCPServer(vk *tables.TableVirtualKey) *server.MCPServer { h.mu.Lock() defer h.mu.Unlock() - vkServer, ok := h.vkMCPServers[vk.Value] + vkServer, ok := h.vkMCPServers[vk.Value.GetValue()] if !ok { // Add new server vkServer = server.NewMCPServer( @@ -298,11 +357,11 @@ func (h *MCPServerHandler) SyncVKMCPServer(vk *tables.TableVirtualKey) *server.M server.WithToolCapabilities(true), ) server.WithToolFilter(h.makeIncludeClientsFilter())(vkServer) - h.vkMCPServers[vk.Value] = vkServer + h.vkMCPServers[vk.Value.GetValue()] = vkServer } availableTools, toolFilter := h.fetchToolsForVK(vk) h.syncServer(vkServer, availableTools, toolFilter) - h.vkMCPServers[vk.Value] = vkServer + h.vkMCPServers[vk.Value.GetValue()] = vkServer logger.Debug("Synced MCP server for virtual key '%s' with %d tools", vk.Name, len(availableTools)) return vkServer } @@ -546,33 +605,251 @@ func (h *MCPServerHandler) makeIncludeClientsFilter() server.ToolFilterFunc { // Utility methods -func (h *MCPServerHandler) getMCPServerForRequest(ctx *fasthttp.RequestCtx) (*server.MCPServer, error) { +// mcpAuthResult carries the outcome of /mcp request authentication. +type mcpAuthResult struct { + mcpServer *server.MCPServer + jwtClaims *jwtMCPClaims // non-nil when authenticated via JWT + jwtVK *tables.TableVirtualKey // non-nil when jwt bf_mode=vk +} + +// getMCPServerForRequest authenticates the /mcp request and returns the +// appropriate scoped MCP server alongside any JWT claims that must be injected +// into the BifrostContext after it is created. +// +// Authentication priority: +// 1. JWT Bearer token (when MCPServerAuthMode is both or oauth) +// 2. VK / header credentials (when MCPServerAuthMode is headers or both) +// 3. Anonymous access (when EnforceAuthOnInference is false) +// +// When MCPServerAuthMode is oauth (strict), header credentials are rejected. +func (h *MCPServerHandler) getMCPServerForRequest(ctx *fasthttp.RequestCtx) (*mcpAuthResult, error) { h.config.Mu.RLock() - enforceVK := h.config.ClientConfig.EnforceAuthOnInference + enforceAuth := h.config.ClientConfig.EnforceAuthOnInference + authMode := h.config.ClientConfig.MCPServerAuthMode h.config.Mu.RUnlock() + discoveryEnabled := authMode == tables.MCPServerAuthModeBoth || authMode == tables.MCPServerAuthModeOAuth + + // --- Pre-authenticated user path --- + // An upstream auth layer that authenticated the caller as a user stamps the + // user id onto the request context. In headers/both modes, scope the request + // to that user's virtual key — the same representative-VK scoping a user-mode + // token gets — so a user authenticated by a bearer token is treated like a + // virtual key. oauth-strict accepts only Bifrost-issued tokens and is excluded. + if h.identityResolver != nil && + (authMode == tables.MCPServerAuthModeHeaders || authMode == tables.MCPServerAuthModeBoth) { + if userID, _ := ctx.UserValue(schemas.BifrostContextKeyUserID).(string); userID != "" { + // The user identity is the sole credential; reject a stray virtual key + // header so it is not also attributed to the request. + if headerVK := getVKFromRequest(ctx); headerVK != "" { + return nil, fmt.Errorf("conflicting credentials: a user token and a virtual key header were both provided; send only one") + } + vkID, err := h.identityResolver.ResolveUserVirtualKey(ctx, userID) + if err != nil { + return nil, err + } + if vkID == "" { + return nil, fmt.Errorf("no MCP access grant for the authenticated user") + } + vk, err := h.getVirtualKeyByID(ctx, vkID) + if err != nil { + return nil, err + } + if !vk.IsActiveValue() { + return nil, fmt.Errorf("virtual key is inactive") + } + vkServer, err := h.ensureVKMCPServerByValue(ctx, vk.Value.GetValue()) + if err != nil { + return nil, err + } + return &mcpAuthResult{mcpServer: vkServer}, nil + } + } + + // --- JWT path --- + if rawJWT := extractBearerJWT(ctx); rawJWT != "" && discoveryEnabled { + // An OAuth token is the sole identity for the request. Reject when a + // header-based virtual key (x-bf-vk / x-api-key / x-goog-api-key / Bearer vk) is also + // presented: mixing credential sources is ambiguous, and for user- and + // session-mode tokens — which carry no virtual key — a stray header VK + // would otherwise leak onto the context and be attributed to the request. + if headerVK := getVKFromRequest(ctx); headerVK != "" { + ctx.Response.Header.Set("WWW-Authenticate", wwwAuthenticateValue(ctx, h.config)) + return nil, fmt.Errorf("conflicting credentials: an OAuth token and a virtual key header were both provided; send only the OAuth token") + } + + // Load the signing key (cached for the process lifetime). A failure here is + // an infrastructure fault — the config store or key is unavailable — not a + // bad token. Log the detail for operators and return a clean message so it + // is never mislabeled as the client's token being invalid. + signingKey, err := h.config.GetOAuth2SigningKey(ctx) + if err != nil { + logger.Error("mcp: failed to load oauth2 signing key for jwt verification: %v", err) + ctx.Response.Header.Set("WWW-Authenticate", wwwAuthenticateValue(ctx, h.config)) + return nil, fmt.Errorf("signing key unavailable") + } + claims, err := verifyMCPJWT(ctx, rawJWT, h.config, signingKey) + if err != nil { + if discoveryEnabled { + ctx.Response.Header.Set("WWW-Authenticate", wwwAuthenticateValue(ctx, h.config)) + } + // Forward verifyMCPJWT's error verbatim: it already labels a genuine + // token failure ("invalid token: ...") precisely, while its config + // faults ("signing key unavailable", ...) must not be mislabeled as + // the client's token being bad. + return nil, err + } + + // For user-mode JWTs, if a dashboard session is present on the request + // (BifrostContextKeyUserID, set by the auth middleware) it must match + // bf_sub — a mismatch means the session and the token disagree on + // identity. Its absence is not fatal: the JWT itself proves identity, and + // initiating a new upstream per-user flow is verified later at the + // session-bearing UI step (flowStart → canAccessUserFlow). + if schemas.MCPAuthMode(claims.BfMode) == schemas.MCPAuthModeUser { + sessionUserID, _ := ctx.UserValue(schemas.BifrostContextKeyUserID).(string) + if sessionUserID != "" && sessionUserID != claims.Subject { + ctx.Response.Header.Set("WWW-Authenticate", wwwAuthenticateValue(ctx, h.config)) + return nil, fmt.Errorf("session user does not match the authenticated token") + } + } + + // Session-mode tokens carry no verified identity. When the operator + // requires authentication (EnforceAuthOnInference=true), session-mode + // JWT requests are rejected — the session itself is not deleted, but + // this endpoint becomes inaccessible until the client re-authenticates + // with a VK or user-mode token. + if schemas.MCPAuthMode(claims.BfMode) == schemas.MCPAuthModeSession && enforceAuth { + ctx.Response.Header.Set("WWW-Authenticate", wwwAuthenticateValue(ctx, h.config)) + return nil, fmt.Errorf("authentication required; session-mode tokens are not accepted when authentication is enforced - re-authenticate with a virtual key or user identity") + } + + res := &mcpAuthResult{jwtClaims: claims} + + // For vk mode, look up the VK by ID to get the scoped server and value. + if schemas.MCPAuthMode(claims.BfMode) == schemas.MCPAuthModeVK { + // Live virtual-key identity cutoff: when virtual-key identity has been + // disabled, reject vk-mode tokens at request time rather than waiting + // for the access token to expire and its refresh to be denied. The + // DisableVKIdentity flag is read first so the common (flag-off) path + // stays a single lock-guarded bool — IsUserModeAvailable is consulted + // only when the flag is set. Gated identically to the refresh cutoff and + // the consent flow's availableModes so it can never fire where vk is + // still an offered authentication path. + if oauth2ServerCfg(h.config).DisableVKIdentity && + h.identityResolver != nil && h.identityResolver.IsUserModeAvailable() { + ctx.Response.Header.Set("WWW-Authenticate", wwwAuthenticateValue(ctx, h.config)) + return nil, fmt.Errorf("virtual-key identity is no longer accepted; re-authenticate") + } + vk, err := h.getVirtualKeyByID(ctx, claims.Subject) + if err != nil { + return nil, err + } + if !vk.IsActiveValue() { + return nil, fmt.Errorf("virtual key is inactive") + } + res.jwtVK = vk + vkServer, serverErr := h.ensureVKMCPServerByValue(ctx, vk.Value.GetValue()) + if serverErr != nil { + return nil, serverErr + } + res.mcpServer = vkServer + } else if scopedServer, scopedErr := h.userScopedServer(ctx, claims); scopedErr != nil { + return nil, scopedErr + } else if scopedServer != nil { + res.mcpServer = scopedServer + } else { + res.mcpServer = h.globalMCPServer + } + return res, nil + } + + // --- oauth strict mode: reject non-JWT requests --- + if authMode == tables.MCPServerAuthModeOAuth { + ctx.Response.Header.Set("WWW-Authenticate", wwwAuthenticateValue(ctx, h.config)) + return nil, fmt.Errorf("this server requires OAuth JWT authentication; header credentials are not accepted in oauth mode") + } + + // --- VK / header credential path --- vk := getVKFromRequest(ctx) // EnforceAuth=false: anonymous access to the global (un-scoped) MCP server // is allowed in dev mode. EnforceAuth=true: VK header is mandatory. - if !enforceVK && vk == "" { - return h.globalMCPServer, nil + if !enforceAuth && vk == "" { + // Anonymous access allowed in dev mode. + return &mcpAuthResult{mcpServer: h.globalMCPServer}, nil } if vk == "" { - return nil, fmt.Errorf("virtual key required to access mcp server; set one of x-bf-vk, Authorization: Bearer , or x-api-key in your MCP client config") + if discoveryEnabled { + ctx.Response.Header.Set("WWW-Authenticate", wwwAuthenticateValue(ctx, h.config)) + } + return nil, fmt.Errorf("virtual key required to access mcp server; set one of x-bf-vk, Authorization: Bearer , x-api-key, or x-goog-api-key in your MCP client config") } - // Fast path: a per-VK server already exists in the cache. + vkServer, err := h.ensureVKMCPServerByValue(ctx, vk) + if err != nil { + return nil, err + } + return &mcpAuthResult{mcpServer: vkServer}, nil +} + +// userScopedServer returns a per-VK MCP server scoped to a user-mode token's +// own tools, or nil (with nil error) when no scoping applies — no resolver, a +// non-user-mode token, or a user with no virtual key — so the caller falls back +// to the global server. +// +// User-mode tokens carry a user identity but no virtual key of their own. The +// resolver maps the user to a representative virtual key (any one of the user's +// equivalent keys), letting this reuse the per-VK scoped server instead of +// serving the global (unscoped) one. Session-mode tokens have no identity to +// scope by and return nil here. +func (h *MCPServerHandler) userScopedServer(ctx *fasthttp.RequestCtx, claims *jwtMCPClaims) (*server.MCPServer, error) { + if h.identityResolver == nil || h.config.ConfigStore == nil { + return nil, nil + } + if schemas.MCPAuthMode(claims.BfMode) != schemas.MCPAuthModeUser { + return nil, nil + } + // Reject deleted or deactivated users at request time, mirroring the vk-mode + // IsActiveValue() cutoff, rather than letting an already-issued access token + // keep working until it expires. Placed before the no-virtual-key early + // return below so a removed user cannot fall through to the global server. + if active, err := h.identityResolver.IsUserActive(ctx, claims.Subject); err != nil { + return nil, fmt.Errorf("failed to verify user: %w", err) + } else if !active { + return nil, fmt.Errorf("user is no longer active") + } + vkID, err := h.identityResolver.ResolveUserVirtualKey(ctx, claims.Subject) + if err != nil { + return nil, fmt.Errorf("failed to resolve virtual key for user: %w", err) + } + if vkID == "" { + return nil, nil + } + // Mirror the vk-mode branch: resolve the representative VK by ID to get its + // value and active state, then reuse the shared per-VK server cache. + vk, err := h.getVirtualKeyByID(ctx, vkID) + if err != nil { + return nil, err + } + if !vk.IsActiveValue() { + return nil, fmt.Errorf("virtual key is inactive") + } + return h.ensureVKMCPServerByValue(ctx, vk.Value.GetValue()) +} + +// ensureVKMCPServerByValue returns the per-VK server from cache or creates it. +func (h *MCPServerHandler) ensureVKMCPServerByValue(ctx context.Context, vkValue string) (*server.MCPServer, error) { h.mu.RLock() - vkServer, ok := h.vkMCPServers[vk] + s, ok := h.vkMCPServers[vkValue] h.mu.RUnlock() + // Fast path: a per-VK server already exists in the cache. if ok { - return vkServer, nil + return s, nil } - // Slow path: build the per-VK server lazily on first use. - return h.ensureVKMCPServer(ctx, vk) + return h.ensureVKMCPServer(ctx, vkValue) } // ensureVKMCPServer lazily builds and caches the MCP server for a virtual key on @@ -588,12 +865,29 @@ func (h *MCPServerHandler) ensureVKMCPServer(ctx context.Context, vkValue string if err != nil || vk == nil { return nil, fmt.Errorf("virtual key not found") } + // GetVirtualKeyByValue does not filter inactive keys, so fail closed here: + // a deactivated key must not yield (or cache) a usable MCP server. This is + // the single chokepoint for both the header path and the JWT vk path. + if !vk.IsActiveValue() { + return nil, fmt.Errorf("virtual key is inactive") + } // SyncVKMCPServer creates (or refreshes) and caches the server under the // handler write lock, returning the live server so a concurrent // SyncAllMCPServers cannot wipe the map out from under us before we read it. return h.SyncVKMCPServer(vk), nil } +// getVKFromRequest extracts a virtual key from the request headers, checking +// each supported header in priority order and returning the first match: +// 1. x-bf-vk — taken verbatim (no prefix check) +// 2. Authorization — "Bearer ", where must start with the VK prefix +// 3. x-api-key — must start with the VK prefix +// 4. x-goog-api-key — must start with the VK prefix +// +// The prefix gate (governance.VirtualKeyPrefix) on the latter three lets real +// provider credentials pass through untouched, so only Bifrost virtual keys are +// picked up here. This header set mirrors the inference path, keeping MCP and +// inference at parity. Returns "" when no header carries a virtual key. func getVKFromRequest(ctx *fasthttp.RequestCtx) string { if value := strings.TrimSpace(string(ctx.Request.Header.Peek(string(schemas.BifrostContextKeyVirtualKey)))); value != "" { return value @@ -615,5 +909,11 @@ func getVKFromRequest(ctx *fasthttp.RequestCtx) string { } } + if googAPIKey := strings.TrimSpace(string(ctx.Request.Header.Peek("x-goog-api-key"))); googAPIKey != "" { + if strings.HasPrefix(strings.ToLower(googAPIKey), governance.VirtualKeyPrefix) { + return googAPIKey + } + } + return "" } diff --git a/transports/bifrost-http/handlers/mcpserver_auth_test.go b/transports/bifrost-http/handlers/mcpserver_auth_test.go new file mode 100644 index 00000000000..224570a5909 --- /dev/null +++ b/transports/bifrost-http/handlers/mcpserver_auth_test.go @@ -0,0 +1,578 @@ +package handlers + +import ( + "testing" + + "github.com/golang-jwt/jwt/v5" + "github.com/mark3labs/mcp-go/server" + "github.com/maximhq/bifrost/core/schemas" + configtables "github.com/maximhq/bifrost/framework/configstore/tables" + "github.com/maximhq/bifrost/transports/bifrost-http/lib" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "github.com/valyala/fasthttp" +) + +// newTestMCPHandler builds an MCPServerHandler around the given config without +// going through NewMCPServerHandler (which needs a live tool manager). Per-VK +// servers are looked up from vkMCPServers; tests pre-seed it to keep the VK +// path from building a real server. +func newTestMCPHandler(cfg *lib.Config) *MCPServerHandler { + return &MCPServerHandler{ + globalMCPServer: server.NewMCPServer("test", "v0", server.WithToolCapabilities(true)), + vkMCPServers: map[string]*server.MCPServer{}, + config: cfg, + } +} + +// TestGetVKFromRequest verifies the VK value is extracted from each supported +// header, in priority order, and that non-VK values are ignored. This mirrors +// the inference-path header set (x-bf-vk, Authorization Bearer, x-api-key, +// x-goog-api-key) so MCP and inference stay at parity. +func TestGetVKFromRequest(t *testing.T) { + const vk = "sk-bf-test-virtual-key" + + cases := []struct { + name string + setup func(*fasthttp.RequestCtx) + wantVK string + }{ + { + name: "x-bf-vk header", + setup: func(ctx *fasthttp.RequestCtx) { ctx.Request.Header.Set("x-bf-vk", vk) }, + wantVK: vk, + }, + { + name: "Authorization Bearer header", + setup: func(ctx *fasthttp.RequestCtx) { ctx.Request.Header.Set("Authorization", "Bearer "+vk) }, + wantVK: vk, + }, + { + name: "x-api-key header", + setup: func(ctx *fasthttp.RequestCtx) { ctx.Request.Header.Set("x-api-key", vk) }, + wantVK: vk, + }, + { + name: "x-goog-api-key header", + setup: func(ctx *fasthttp.RequestCtx) { ctx.Request.Header.Set("x-goog-api-key", vk) }, + wantVK: vk, + }, + { + name: "no header returns empty string", + setup: func(*fasthttp.RequestCtx) {}, + wantVK: "", + }, + { + name: "non-VK Bearer token returns empty string", + setup: func(ctx *fasthttp.RequestCtx) { ctx.Request.Header.Set("Authorization", "Bearer regular-api-key-123") }, + wantVK: "", + }, + { + name: "non-VK x-goog-api-key returns empty string", + setup: func(ctx *fasthttp.RequestCtx) { ctx.Request.Header.Set("x-goog-api-key", "regular-google-key") }, + wantVK: "", + }, + { + name: "x-bf-vk takes priority over x-goog-api-key", + setup: func(ctx *fasthttp.RequestCtx) { + ctx.Request.Header.Set("x-bf-vk", vk) + ctx.Request.Header.Set("x-goog-api-key", "sk-bf-other") + }, + wantVK: vk, + }, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + ctx := &fasthttp.RequestCtx{} + tc.setup(ctx) + assert.Equal(t, tc.wantVK, getVKFromRequest(ctx)) + }) + } +} + +// TestGetMCPServerForRequest_JWTPath covers the JWT branch of /mcp auth across +// modes and identity kinds: the security contract for OAuth-authenticated calls. +func TestGetMCPServerForRequest_JWTPath(t *testing.T) { + SetLogger(&mockLogger{}) + key, priv := newTestSigningKey(t) + + t.Run("oauth mode: valid vk JWT with active key is accepted", func(t *testing.T) { + activeVK := &configtables.TableVirtualKey{ID: "vk-row-1", Value: *schemas.NewSecretVar("sk-bf-active"), IsActive: new(true)} + store := &mockOAuth2Store{ + signingKey: key, + vksByID: map[string]*configtables.TableVirtualKey{"vk-row-1": activeVK}, + } + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeOAuth, false) + h := newTestMCPHandler(cfg) + // Pre-seed the per-VK server so the accepted path does not build one. + h.vkMCPServers[activeVK.Value.GetValue()] = server.NewMCPServer("vk", "v0") + + raw := mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["bf_mode"] = string(schemas.MCPAuthModeVK) + c["sub"] = "vk-row-1" + }) + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("Authorization", "Bearer "+raw) + + res, err := h.getMCPServerForRequest(ctx) + require.NoError(t, err) + require.NotNil(t, res) + require.NotNil(t, res.jwtClaims) + require.NotNil(t, res.jwtVK) + assert.Equal(t, "vk-row-1", res.jwtVK.ID) + assert.NotNil(t, res.mcpServer) + }) + + t.Run("vk JWT with inactive key is rejected", func(t *testing.T) { + inactiveVK := &configtables.TableVirtualKey{ID: "vk-row-1", Value: *schemas.NewSecretVar("sk-bf-x"), IsActive: new(false)} + store := &mockOAuth2Store{ + signingKey: key, + vksByID: map[string]*configtables.TableVirtualKey{"vk-row-1": inactiveVK}, + } + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + h := newTestMCPHandler(cfg) + + raw := mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["bf_mode"] = string(schemas.MCPAuthModeVK) + c["sub"] = "vk-row-1" + }) + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("Authorization", "Bearer "+raw) + + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "inactive") + }) + + t.Run("vk JWT for unknown key is rejected", func(t *testing.T) { + store := &mockOAuth2Store{signingKey: key, vksByID: map[string]*configtables.TableVirtualKey{}} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + h := newTestMCPHandler(cfg) + + raw := mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["bf_mode"] = string(schemas.MCPAuthModeVK) + c["sub"] = "missing-vk" + }) + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("Authorization", "Bearer "+raw) + + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + }) + + t.Run("user JWT without a session is allowed", func(t *testing.T) { + store := &mockOAuth2Store{signingKey: key} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + h := newTestMCPHandler(cfg) + + raw := mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["bf_mode"] = string(schemas.MCPAuthModeUser) + c["sub"] = "user-1" + }) + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("Authorization", "Bearer "+raw) + + res, err := h.getMCPServerForRequest(ctx) + require.NoError(t, err) + require.NotNil(t, res) + // No resolver wired → falls back to the global server. + assert.Equal(t, h.globalMCPServer, res.mcpServer) + }) + + t.Run("user JWT is scoped to the user's representative virtual key", func(t *testing.T) { + activeVK := &configtables.TableVirtualKey{ID: "vk-row-1", Value: *schemas.NewSecretVar("sk-bf-user-rep"), IsActive: new(true)} + store := &mockOAuth2Store{ + signingKey: key, + vksByID: map[string]*configtables.TableVirtualKey{"vk-row-1": activeVK}, + } + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + h := newTestMCPHandler(cfg) + // Resolver maps the user to a representative VK; pre-seed its server so the + // scoped path does not build a real one. + h.identityResolver = &fakeResolver{userVKID: "vk-row-1"} + vkServer := server.NewMCPServer("vk", "v0") + h.vkMCPServers[activeVK.Value.GetValue()] = vkServer + + raw := mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["bf_mode"] = string(schemas.MCPAuthModeUser) + c["sub"] = "user-1" + }) + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("Authorization", "Bearer "+raw) + + res, err := h.getMCPServerForRequest(ctx) + require.NoError(t, err) + require.NotNil(t, res) + // Served the user's scoped VK server, NOT the global (unscoped) server. + assert.Equal(t, vkServer, res.mcpServer) + assert.NotEqual(t, h.globalMCPServer, res.mcpServer) + // User mode keeps the user identity — no VK identity is attributed. + assert.Nil(t, res.jwtVK) + }) + + t.Run("user JWT is rejected when the user is no longer active", func(t *testing.T) { + activeVK := &configtables.TableVirtualKey{ID: "vk-row-1", Value: *schemas.NewSecretVar("sk-bf-user-rep"), IsActive: new(true)} + store := &mockOAuth2Store{ + signingKey: key, + vksByID: map[string]*configtables.TableVirtualKey{"vk-row-1": activeVK}, + } + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + h := newTestMCPHandler(cfg) + // The resolver still maps the user to a VK, but reports the user as gone. + // The request must be rejected at request time rather than falling through + // to the global (unscoped) server until the access token expires. + h.identityResolver = &fakeResolver{userVKID: "vk-row-1", userInactive: true} + + raw := mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["bf_mode"] = string(schemas.MCPAuthModeUser) + c["sub"] = "user-1" + }) + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("Authorization", "Bearer "+raw) + + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + }) + + t.Run("user JWT falls back to the global server when the user has no virtual key", func(t *testing.T) { + store := &mockOAuth2Store{signingKey: key} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + h := newTestMCPHandler(cfg) + h.identityResolver = &fakeResolver{userVKID: ""} // user has no AP-managed VK + + raw := mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["bf_mode"] = string(schemas.MCPAuthModeUser) + c["sub"] = "user-1" + }) + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("Authorization", "Bearer "+raw) + + res, err := h.getMCPServerForRequest(ctx) + require.NoError(t, err) + assert.Equal(t, h.globalMCPServer, res.mcpServer) + }) + + t.Run("user JWT with a matching session is accepted", func(t *testing.T) { + store := &mockOAuth2Store{signingKey: key} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + h := newTestMCPHandler(cfg) + + raw := mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["bf_mode"] = string(schemas.MCPAuthModeUser) + c["sub"] = "user-1" + }) + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("Authorization", "Bearer "+raw) + ctx.SetUserValue(schemas.BifrostContextKeyUserID, "user-1") + + _, err := h.getMCPServerForRequest(ctx) + require.NoError(t, err) + }) + + t.Run("user JWT with a mismatched session is rejected", func(t *testing.T) { + store := &mockOAuth2Store{signingKey: key} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + h := newTestMCPHandler(cfg) + + raw := mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["bf_mode"] = string(schemas.MCPAuthModeUser) + c["sub"] = "user-1" + }) + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("Authorization", "Bearer "+raw) + ctx.SetUserValue(schemas.BifrostContextKeyUserID, "someone-else") + + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + }) + + t.Run("session JWT is rejected when auth is enforced", func(t *testing.T) { + store := &mockOAuth2Store{signingKey: key} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, true) + h := newTestMCPHandler(cfg) + + raw := mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["bf_mode"] = string(schemas.MCPAuthModeSession) + c["sub"] = "session-abc" + }) + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("Authorization", "Bearer "+raw) + + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + }) + + t.Run("session JWT is accepted when auth is not enforced", func(t *testing.T) { + store := &mockOAuth2Store{signingKey: key} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + h := newTestMCPHandler(cfg) + + raw := mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["bf_mode"] = string(schemas.MCPAuthModeSession) + c["sub"] = "session-abc" + }) + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("Authorization", "Bearer "+raw) + + res, err := h.getMCPServerForRequest(ctx) + require.NoError(t, err) + assert.Equal(t, h.globalMCPServer, res.mcpServer) + }) + + t.Run("both mode: session token with a header VK is rejected", func(t *testing.T) { + store := &mockOAuth2Store{signingKey: key} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + h := newTestMCPHandler(cfg) + + raw := mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["bf_mode"] = string(schemas.MCPAuthModeSession) + c["sub"] = "session-abc" + }) + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("Authorization", "Bearer "+raw) + ctx.Request.Header.Set(string(schemas.BifrostContextKeyVirtualKey), "sk-bf-header") + + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "conflicting credentials") + }) + + t.Run("both mode: vk token with a header VK is rejected", func(t *testing.T) { + store := &mockOAuth2Store{signingKey: key} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, false) + h := newTestMCPHandler(cfg) + + raw := mintTestToken(t, priv, key.KID, func(c jwt.MapClaims) { + c["bf_mode"] = string(schemas.MCPAuthModeVK) + c["sub"] = "vk-row-1" + }) + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("Authorization", "Bearer "+raw) + ctx.Request.Header.Set(string(schemas.BifrostContextKeyVirtualKey), "sk-bf-header") + + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "conflicting credentials") + }) +} + +// TestGetMCPServerForRequest_PreAuthenticatedUserPath covers the path where an +// upstream auth layer has already authenticated the caller as a user and stamped +// the user id onto the request context (BifrostContextKeyUserID). In headers/both +// modes the request is scoped to the user's representative virtual key, just like +// a user-mode token; oauth-strict ignores it (Bifrost-issued tokens only). +func TestGetMCPServerForRequest_PreAuthenticatedUserPath(t *testing.T) { + SetLogger(&mockLogger{}) + key, _ := newTestSigningKey(t) + + activeVK := &configtables.TableVirtualKey{ID: "vk-row-1", Value: *schemas.NewSecretVar("sk-bf-user-rep"), IsActive: new(true)} + newStore := func() *mockOAuth2Store { + return &mockOAuth2Store{ + signingKey: key, + vksByID: map[string]*configtables.TableVirtualKey{"vk-row-1": activeVK}, + } + } + + for _, mode := range []configtables.MCPServerAuthMode{ + configtables.MCPServerAuthModeHeaders, + configtables.MCPServerAuthModeBoth, + } { + t.Run(string(mode)+" mode: stamped user id is scoped to the user's virtual key", func(t *testing.T) { + cfg := newTestOAuth2Config(newStore(), mode, true) + h := newTestMCPHandler(cfg) + h.identityResolver = &fakeResolver{userVKID: "vk-row-1"} + vkServer := server.NewMCPServer("vk", "v0") + h.vkMCPServers[activeVK.Value.GetValue()] = vkServer + + ctx := &fasthttp.RequestCtx{} + ctx.SetUserValue(schemas.BifrostContextKeyUserID, "user-1") + + res, err := h.getMCPServerForRequest(ctx) + require.NoError(t, err) + require.NotNil(t, res) + // Served the user's scoped VK server, NOT the global (unscoped) one. + assert.Equal(t, vkServer, res.mcpServer) + assert.NotEqual(t, h.globalMCPServer, res.mcpServer) + // Identity stays the user; no VK identity or JWT claims are attributed. + assert.Nil(t, res.jwtVK) + assert.Nil(t, res.jwtClaims) + }) + } + + t.Run("user with no virtual key is rejected (strict VK parity)", func(t *testing.T) { + cfg := newTestOAuth2Config(newStore(), configtables.MCPServerAuthModeBoth, true) + h := newTestMCPHandler(cfg) + h.identityResolver = &fakeResolver{userVKID: ""} // no AP-managed VK + + ctx := &fasthttp.RequestCtx{} + ctx.SetUserValue(schemas.BifrostContextKeyUserID, "user-1") + + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "no MCP access grant") + }) + + t.Run("stamped user id with a header VK is rejected as conflicting", func(t *testing.T) { + cfg := newTestOAuth2Config(newStore(), configtables.MCPServerAuthModeBoth, true) + h := newTestMCPHandler(cfg) + h.identityResolver = &fakeResolver{userVKID: "vk-row-1"} + h.vkMCPServers[activeVK.Value.GetValue()] = server.NewMCPServer("vk", "v0") + + ctx := &fasthttp.RequestCtx{} + ctx.SetUserValue(schemas.BifrostContextKeyUserID, "user-1") + ctx.Request.Header.Set(string(schemas.BifrostContextKeyVirtualKey), "sk-bf-header") + + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "conflicting credentials") + }) + + t.Run("inactive representative virtual key is rejected", func(t *testing.T) { + inactiveVK := &configtables.TableVirtualKey{ID: "vk-row-1", Value: *schemas.NewSecretVar("sk-bf-x"), IsActive: new(false)} + store := &mockOAuth2Store{ + signingKey: key, + vksByID: map[string]*configtables.TableVirtualKey{"vk-row-1": inactiveVK}, + } + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeBoth, true) + h := newTestMCPHandler(cfg) + h.identityResolver = &fakeResolver{userVKID: "vk-row-1"} + + ctx := &fasthttp.RequestCtx{} + ctx.SetUserValue(schemas.BifrostContextKeyUserID, "user-1") + + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "inactive") + }) + + t.Run("oauth-strict mode ignores the stamped user id", func(t *testing.T) { + cfg := newTestOAuth2Config(newStore(), configtables.MCPServerAuthModeOAuth, true) + h := newTestMCPHandler(cfg) + h.identityResolver = &fakeResolver{userVKID: "vk-row-1"} + + ctx := &fasthttp.RequestCtx{} + ctx.SetUserValue(schemas.BifrostContextKeyUserID, "user-1") + + // No bearer JWT, and header credentials are rejected in oauth-strict: the + // user-id check does not run, so this falls through to the strict rejection. + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "OAuth JWT") + }) + + t.Run("no resolver: stamped user id is ignored", func(t *testing.T) { + cfg := newTestOAuth2Config(newStore(), configtables.MCPServerAuthModeHeaders, false) + h := newTestMCPHandler(cfg) + // identityResolver is nil (pure OSS, no IdP); enforce=false → anonymous. + + ctx := &fasthttp.RequestCtx{} + ctx.SetUserValue(schemas.BifrostContextKeyUserID, "user-1") + + res, err := h.getMCPServerForRequest(ctx) + require.NoError(t, err) + assert.Equal(t, h.globalMCPServer, res.mcpServer) + }) +} + +// TestGetMCPServerForRequest_HeaderAndAnonPath covers the legacy header-VK path, +// anonymous access, and oauth-strict header rejection. +func TestGetMCPServerForRequest_HeaderAndAnonPath(t *testing.T) { + SetLogger(&mockLogger{}) + key, _ := newTestSigningKey(t) + + t.Run("headers mode: active header VK connects", func(t *testing.T) { + activeVK := &configtables.TableVirtualKey{ID: "vk-row-1", Value: *schemas.NewSecretVar("sk-bf-active"), IsActive: new(true)} + store := &mockOAuth2Store{ + signingKey: key, + vksByValue: map[string]*configtables.TableVirtualKey{"sk-bf-active": activeVK}, + } + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeHeaders, true) + h := newTestMCPHandler(cfg) + h.vkMCPServers[activeVK.Value.GetValue()] = server.NewMCPServer("vk", "v0") + + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set(string(schemas.BifrostContextKeyVirtualKey), "sk-bf-active") + + res, err := h.getMCPServerForRequest(ctx) + require.NoError(t, err) + assert.Nil(t, res.jwtClaims) + assert.NotNil(t, res.mcpServer) + }) + + t.Run("inactive header VK is rejected at the shared chokepoint", func(t *testing.T) { + inactiveVK := &configtables.TableVirtualKey{ID: "vk-row-1", Value: *schemas.NewSecretVar("sk-bf-x"), IsActive: new(false)} + store := &mockOAuth2Store{ + signingKey: key, + vksByValue: map[string]*configtables.TableVirtualKey{"sk-bf-x": inactiveVK}, + } + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeHeaders, true) + h := newTestMCPHandler(cfg) // not pre-seeded: must traverse ensureVKMCPServer + + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set(string(schemas.BifrostContextKeyVirtualKey), "sk-bf-x") + + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "inactive") + }) + + t.Run("anonymous access yields the global server when auth is not enforced", func(t *testing.T) { + store := &mockOAuth2Store{signingKey: key} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeHeaders, false) + h := newTestMCPHandler(cfg) + + ctx := &fasthttp.RequestCtx{} + res, err := h.getMCPServerForRequest(ctx) + require.NoError(t, err) + assert.Equal(t, h.globalMCPServer, res.mcpServer) + }) + + t.Run("no credentials rejected when auth is enforced", func(t *testing.T) { + store := &mockOAuth2Store{signingKey: key} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeHeaders, true) + h := newTestMCPHandler(cfg) + + ctx := &fasthttp.RequestCtx{} + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + }) + + t.Run("oauth strict mode rejects a header VK with WWW-Authenticate", func(t *testing.T) { + store := &mockOAuth2Store{signingKey: key} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeOAuth, false) + h := newTestMCPHandler(cfg) + + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set(string(schemas.BifrostContextKeyVirtualKey), "sk-bf-active") + + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + assert.NotEmpty(t, ctx.Response.Header.Peek("WWW-Authenticate")) + }) + + t.Run("oauth strict mode with no credentials sets WWW-Authenticate", func(t *testing.T) { + store := &mockOAuth2Store{signingKey: key} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeOAuth, false) + h := newTestMCPHandler(cfg) + + ctx := &fasthttp.RequestCtx{} + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + assert.NotEmpty(t, ctx.Response.Header.Peek("WWW-Authenticate")) + }) + + t.Run("headers mode: a JWT bearer is not treated as a credential", func(t *testing.T) { + store := &mockOAuth2Store{signingKey: key} + cfg := newTestOAuth2Config(store, configtables.MCPServerAuthModeHeaders, true) + h := newTestMCPHandler(cfg) + + // A JWT-looking bearer in headers mode: discovery is off, so the JWT path + // is skipped and the token is not a VK — with auth enforced this rejects. + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.Set("Authorization", "Bearer eyJhbGciOiJSUzI1NiJ9.payload.sig") + + _, err := h.getMCPServerForRequest(ctx) + require.Error(t, err) + }) +} diff --git a/transports/bifrost-http/handlers/mcp_sessions.go b/transports/bifrost-http/handlers/mcpsessions.go similarity index 99% rename from transports/bifrost-http/handlers/mcp_sessions.go rename to transports/bifrost-http/handlers/mcpsessions.go index 3fb70f55ee8..34882c8534b 100644 --- a/transports/bifrost-http/handlers/mcp_sessions.go +++ b/transports/bifrost-http/handlers/mcpsessions.go @@ -125,6 +125,10 @@ func parseMCPSessionsListQuery(ctx *fasthttp.RequestCtx) (mcpSessionsListQuery, q := mcpSessionsListQuery{Limit: mcpSessionsDefaultLimit} args := ctx.QueryArgs() q.Filters.Search = strings.TrimSpace(string(args.Peek("q"))) + // Identity is an exact-match SQL key (bf_sub), unlike the fuzzy Search term — + // pass it through verbatim so identities with significant leading/trailing + // whitespace still resolve. + q.Filters.Identity = string(args.Peek("identity")) q.Filters.Statuses = parseCommaSeparated(string(args.Peek("status"))) q.Filters.AuthModes = parseCommaSeparated(string(args.Peek("auth_mode"))) q.Filters.MCPClientIDs = parseCommaSeparated(string(args.Peek("mcp_client_id"))) diff --git a/transports/bifrost-http/handlers/mcp_sessions_test.go b/transports/bifrost-http/handlers/mcpsessions_test.go similarity index 100% rename from transports/bifrost-http/handlers/mcp_sessions_test.go rename to transports/bifrost-http/handlers/mcpsessions_test.go diff --git a/transports/bifrost-http/handlers/middlewares.go b/transports/bifrost-http/handlers/middlewares.go index 20bb0a8bb70..2b8bdf212c0 100644 --- a/transports/bifrost-http/handlers/middlewares.go +++ b/transports/bifrost-http/handlers/middlewares.go @@ -916,6 +916,11 @@ func (m *AuthMiddleware) APIMiddleware() schemas.BifrostHTTPMiddleware { // credentials securely. Management endpoints under /api/skills (without // /serve/) remain authenticated. "/api/skills/serve/", + // OAuth2 discovery endpoints (RFC 8414 AS metadata, RFC 9728 protected + // resource metadata, RFC 7517 JWKS) must be reachable without auth so + // clients can bootstrap the flow. Each handler still gates availability + // behind discoveryEnabled() and serves 404 when OAuth mode is off. + "/.well-known/", } return m.middleware(func(authConfig *configstore.AuthConfig, url string) bool { if slices.Contains(systemWhitelistedRoutes, url) || diff --git a/transports/bifrost-http/handlers/plugins_test.go b/transports/bifrost-http/handlers/plugins_test.go index 0b0dc287914..f0f4f7d3333 100644 --- a/transports/bifrost-http/handlers/plugins_test.go +++ b/transports/bifrost-http/handlers/plugins_test.go @@ -55,6 +55,7 @@ func (noopPluginsLoader) GetLoadedPluginNames() []string { return nil } func (noopPluginsLoader) NormalizePluginConfig(_ string, _ map[string]any) (map[string]any, error) { return nil, nil } + func (noopPluginsLoader) ExpandPluginConfigForAPI(_ string, _ map[string]any) (map[string]any, error) { return nil, nil } @@ -127,7 +128,7 @@ func TestRestoreRedacted_OTELProfilesHeaders(t *testing.T) { } // An intentional env.* reference (e.g. credential rotation) must pass through. - // NewEnvVar parses the "env." prefix as FromEnv=true, which IsRedacted reports as + // NewSecretVar parses the "env." prefix as FromEnv=true, which IsRedacted reports as // redacted; the IsFromEnv guard must let it through rather than restoring the stored value. rotated := mkConfig("env.NEW_TOKEN", "env.NEW_VERSION") got3 := restoreRedactedFromExisting(rotated, existing) diff --git a/transports/bifrost-http/handlers/provider_keys.go b/transports/bifrost-http/handlers/provider_keys.go index 4869c6ac9b0..382a60cf53d 100644 --- a/transports/bifrost-http/handlers/provider_keys.go +++ b/transports/bifrost-http/handlers/provider_keys.go @@ -74,7 +74,7 @@ func (h *ProviderHandler) createProviderKey(ctx *fasthttp.RequestCtx) { var key schemas.Key if err := sonic.Unmarshal(ctx.PostBody(), &key); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid JSON: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } @@ -169,7 +169,7 @@ func (h *ProviderHandler) updateProviderKey(ctx *fasthttp.RequestCtx) { var updateKey schemas.Key if err := sonic.Unmarshal(ctx.PostBody(), &updateKey); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid JSON: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } @@ -429,6 +429,47 @@ func (h *ProviderHandler) mergeUpdatedKey(oldRawKey, oldRedactedKey, updateKey s } } + if updateKey.BedrockMantleKeyConfig != nil && oldRedactedKey.BedrockMantleKeyConfig != nil && oldRawKey.BedrockMantleKeyConfig != nil { + if updateKey.BedrockMantleKeyConfig.AccessKey.IsRedacted() && + updateKey.BedrockMantleKeyConfig.AccessKey.Equals(&oldRedactedKey.BedrockMantleKeyConfig.AccessKey) { + mergedKey.BedrockMantleKeyConfig.AccessKey = oldRawKey.BedrockMantleKeyConfig.AccessKey + } + if updateKey.BedrockMantleKeyConfig.SecretKey.IsRedacted() && + updateKey.BedrockMantleKeyConfig.SecretKey.Equals(&oldRedactedKey.BedrockMantleKeyConfig.SecretKey) { + mergedKey.BedrockMantleKeyConfig.SecretKey = oldRawKey.BedrockMantleKeyConfig.SecretKey + } + if updateKey.BedrockMantleKeyConfig.SessionToken != nil && + oldRedactedKey.BedrockMantleKeyConfig.SessionToken != nil && + updateKey.BedrockMantleKeyConfig.SessionToken.IsRedacted() && + updateKey.BedrockMantleKeyConfig.SessionToken.Equals(oldRedactedKey.BedrockMantleKeyConfig.SessionToken) { + mergedKey.BedrockMantleKeyConfig.SessionToken = oldRawKey.BedrockMantleKeyConfig.SessionToken + } + if updateKey.BedrockMantleKeyConfig.Region != nil && + oldRedactedKey.BedrockMantleKeyConfig.Region != nil && + updateKey.BedrockMantleKeyConfig.Region.IsRedacted() && + updateKey.BedrockMantleKeyConfig.Region.Equals(oldRedactedKey.BedrockMantleKeyConfig.Region) { + mergedKey.BedrockMantleKeyConfig.Region = oldRawKey.BedrockMantleKeyConfig.Region + } + if updateKey.BedrockMantleKeyConfig.RoleARN != nil && + oldRedactedKey.BedrockMantleKeyConfig.RoleARN != nil && + updateKey.BedrockMantleKeyConfig.RoleARN.IsRedacted() && + updateKey.BedrockMantleKeyConfig.RoleARN.Equals(oldRedactedKey.BedrockMantleKeyConfig.RoleARN) { + mergedKey.BedrockMantleKeyConfig.RoleARN = oldRawKey.BedrockMantleKeyConfig.RoleARN + } + if updateKey.BedrockMantleKeyConfig.ExternalID != nil && + oldRedactedKey.BedrockMantleKeyConfig.ExternalID != nil && + updateKey.BedrockMantleKeyConfig.ExternalID.IsRedacted() && + updateKey.BedrockMantleKeyConfig.ExternalID.Equals(oldRedactedKey.BedrockMantleKeyConfig.ExternalID) { + mergedKey.BedrockMantleKeyConfig.ExternalID = oldRawKey.BedrockMantleKeyConfig.ExternalID + } + if updateKey.BedrockMantleKeyConfig.RoleSessionName != nil && + oldRedactedKey.BedrockMantleKeyConfig.RoleSessionName != nil && + updateKey.BedrockMantleKeyConfig.RoleSessionName.IsRedacted() && + updateKey.BedrockMantleKeyConfig.RoleSessionName.Equals(oldRedactedKey.BedrockMantleKeyConfig.RoleSessionName) { + mergedKey.BedrockMantleKeyConfig.RoleSessionName = oldRawKey.BedrockMantleKeyConfig.RoleSessionName + } + } + if updateKey.VLLMKeyConfig != nil && oldRedactedKey.VLLMKeyConfig != nil && oldRawKey.VLLMKeyConfig != nil { if updateKey.VLLMKeyConfig.URL.IsRedacted() && updateKey.VLLMKeyConfig.URL.Equals(&oldRedactedKey.VLLMKeyConfig.URL) { diff --git a/transports/bifrost-http/handlers/providers.go b/transports/bifrost-http/handlers/providers.go index d2d10023bf0..16196539682 100644 --- a/transports/bifrost-http/handlers/providers.go +++ b/transports/bifrost-http/handlers/providers.go @@ -240,7 +240,7 @@ func (h *ProviderHandler) getProvider(ctx *fasthttp.RequestCtx) { func (h *ProviderHandler) addProvider(ctx *fasthttp.RequestCtx) { var payload providerCreatePayload if err := sonic.Unmarshal(ctx.PostBody(), &payload); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid JSON: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } // Validate provider @@ -385,7 +385,7 @@ func (h *ProviderHandler) updateProvider(ctx *fasthttp.RequestCtx) { }{} if err := sonic.Unmarshal(ctx.PostBody(), &payload); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid JSON: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } @@ -624,6 +624,7 @@ func (h *ProviderHandler) listKeys(ctx *fasthttp.RequestCtx) { type ModelResponse struct { Name string `json:"name"` Provider string `json:"provider"` + IsDeprecated bool `json:"is_deprecated,omitempty"` AccessibleByKeys []string `json:"accessible_by_keys,omitempty"` } @@ -641,6 +642,7 @@ type ModelDetailsResponse struct { MaxInputTokens *int `json:"max_input_tokens,omitempty"` MaxOutputTokens *int `json:"max_output_tokens,omitempty"` Architecture *schemas.Architecture `json:"architecture,omitempty"` + IsDeprecated bool `json:"is_deprecated,omitempty"` AdditionalAttributes map[string]string `json:"additional_attributes,omitempty"` AccessibleByKeys []string `json:"accessible_by_keys,omitempty"` } @@ -694,8 +696,9 @@ func (h *ProviderHandler) listModels(ctx *fasthttp.RequestCtx) { responseModels := make([]ModelResponse, 0, len(allModels)) for _, model := range allModels { entry := ModelResponse{ - Name: model.Name, - Provider: string(model.Provider), + Name: model.Name, + Provider: string(model.Provider), + IsDeprecated: h.isModelDeprecated(model.Name, model.Provider), } if len(model.AccessibleByKeys) > 0 { entry.AccessibleByKeys = model.AccessibleByKeys @@ -755,6 +758,7 @@ func (h *ProviderHandler) listModelDetails(ctx *fasthttp.RequestCtx) { details.MaxInputTokens = capabilities.MaxInputTokens details.MaxOutputTokens = capabilities.MaxOutputTokens details.Architecture = capabilities.Architecture + details.IsDeprecated = capabilities.IsDeprecated details.AdditionalAttributes = capabilities.AdditionalAttributes } responseModels = append(responseModels, details) @@ -766,6 +770,15 @@ func (h *ProviderHandler) listModelDetails(ctx *fasthttp.RequestCtx) { }) } +func (h *ProviderHandler) isModelDeprecated(model string, provider schemas.ModelProvider) bool { + modelCatalog := h.inMemoryStore.ModelCatalog + if modelCatalog == nil { + return false + } + capabilities := modelCatalog.GetModelCapabilityEntryForModel(model, provider) + return capabilities != nil && capabilities.IsDeprecated +} + // parseModelListQuery normalizes the management model-list query string and resolves // any virtual key present in the request headers to populate provider/model filters. func (h *ProviderHandler) parseModelListQuery(ctx *fasthttp.RequestCtx, defaultLimit int) (modelListQuery, bool) { @@ -1261,7 +1274,7 @@ func validateRetryBackoff(networkConfig *schemas.NetworkConfig) error { func (h *ProviderHandler) upsertModelCatalogEntries(ctx *fasthttp.RequestCtx) { var payload []ModelPricingAttributesEntry if err := sonic.Unmarshal(ctx.PostBody(), &payload); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid JSON: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } for i := range payload { diff --git a/transports/bifrost-http/handlers/providers_test.go b/transports/bifrost-http/handlers/providers_test.go index 4e11ecb1f0b..8b1eea07ca1 100644 --- a/transports/bifrost-http/handlers/providers_test.go +++ b/transports/bifrost-http/handlers/providers_test.go @@ -3,6 +3,9 @@ package handlers import ( "context" "encoding/json" + "os" + "path/filepath" + "slices" "strings" "testing" @@ -11,6 +14,7 @@ import ( "github.com/maximhq/bifrost/framework/configstore" configstoreTables "github.com/maximhq/bifrost/framework/configstore/tables" "github.com/maximhq/bifrost/framework/modelcatalog" + "github.com/maximhq/bifrost/framework/modelcatalog/datasheet" governanceplugin "github.com/maximhq/bifrost/plugins/governance" "github.com/maximhq/bifrost/transports/bifrost-http/lib" "github.com/valyala/fasthttp" @@ -344,9 +348,17 @@ func TestUpdateProvider_PassesThroughForEmptyOrAbsentKeys(t *testing.T) { } } -// boolPtr keeps pointer-valued key fixtures inline without pulling in pointer helpers. -func boolPtr(v bool) *bool { - return &v +func modelCatalogForPricingJSON(t *testing.T, pricingJSON []byte) *modelcatalog.ModelCatalog { + t.Helper() + pricingPath := filepath.Join(t.TempDir(), "pricing.json") + if err := os.WriteFile(pricingPath, pricingJSON, 0o600); err != nil { + t.Fatalf("write pricing testdata: %v", err) + } + ds := datasheet.New(nil, nil, datasheet.Config{URL: "file://" + pricingPath}) + if err := ds.LoadFromURLIntoMemory(t.Context()); err != nil { + t.Fatalf("load pricing testdata: %v", err) + } + return modelcatalog.NewTestCatalogWithDatasheet(ds) } func TestListModels_UnknownKeysDoNotFilter(t *testing.T) { @@ -395,7 +407,7 @@ func TestListModels_ReturnsExactAccessibleByKeysAndSkipsDisabledKeys(t *testing. []schemas.Key{ {ID: "key-a", Models: []string{"gpt-4o"}}, {ID: "key-b", Models: []string{"gpt-4o", "gpt-4o-mini"}}, - {ID: "key-disabled", Enabled: boolPtr(false)}, + {ID: "key-disabled", Enabled: new(false)}, }, []string{"gpt-4o", "gpt-4o-mini"}, []string{"gpt-4o", "gpt-4o-mini"}, @@ -469,6 +481,119 @@ func TestListModels_AppliesQueryAndLimitAfterFiltering(t *testing.T) { } } +func TestListModels_MarksDeprecatedModelsWithoutFiltering(t *testing.T) { + SetLogger(&mockLogger{}) + + h := providerHandlerForTest( + schemas.OpenAI, + []schemas.Key{{ID: "key-a"}}, + []string{"deprecated-model", "current-model", "another-current-model"}, + []string{"deprecated-model", "current-model", "another-current-model"}, + ) + + pricingJSON := []byte(`{ + "deprecated-model": {"provider":"openai","mode":"chat","base_model":"deprecated-model","is_deprecated":true}, + "current-model": {"provider":"openai","mode":"chat","base_model":"current-model"}, + "another-current-model": {"provider":"openai","mode":"chat","base_model":"another-current-model"} + }`) + h.inMemoryStore.ModelCatalog = modelCatalogForPricingJSON(t, pricingJSON) + + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.SetMethod("GET") + ctx.Request.SetRequestURI("/api/models?provider=openai&limit=10") + + h.listModels(ctx) + + if ctx.Response.StatusCode() != fasthttp.StatusOK { + t.Fatalf("expected 200, got %d: %s", ctx.Response.StatusCode(), string(ctx.Response.Body())) + } + + var resp ListModelsResponse + if err := json.Unmarshal(ctx.Response.Body(), &resp); err != nil { + t.Fatalf("failed to unmarshal response: %v", err) + } + + if resp.Total != 3 { + t.Fatalf("expected total=3 (deprecated models are not filtered), got %d", resp.Total) + } + var deprecated *ModelResponse + for i := range resp.Models { + if resp.Models[i].Name == "deprecated-model" { + deprecated = &resp.Models[i] + } + } + if deprecated == nil { + t.Fatalf("deprecated model should still be returned, got %#v", resp.Models) + } + if !deprecated.IsDeprecated { + t.Fatalf("deprecated model should carry is_deprecated=true, got %#v", *deprecated) + } +} + +func TestListBaseModels_IncludesDeprecatedPricingRows(t *testing.T) { + SetLogger(&mockLogger{}) + + h := providerHandlerForTest( + schemas.OpenAI, + []schemas.Key{{ID: "key-a"}}, + nil, + nil, + ) + h.inMemoryStore.ModelCatalog = modelCatalogForPricingJSON(t, []byte(`{ + "deprecated-model": {"provider":"openai","mode":"chat","base_model":"deprecated-base","is_deprecated":true}, + "current-model": {"provider":"openai","mode":"chat","base_model":"current-base"} + }`)) + + ctx := &fasthttp.RequestCtx{} + ctx.Request.Header.SetMethod("GET") + ctx.Request.SetRequestURI("/api/models/base?limit=10") + + h.listBaseModels(ctx) + + if ctx.Response.StatusCode() != fasthttp.StatusOK { + t.Fatalf("expected 200, got %d: %s", ctx.Response.StatusCode(), string(ctx.Response.Body())) + } + + var resp ListBaseModelsResponse + if err := json.Unmarshal(ctx.Response.Body(), &resp); err != nil { + t.Fatalf("failed to unmarshal response: %v", err) + } + if resp.Total != 2 || !slices.Contains(resp.Models, "current-base") || !slices.Contains(resp.Models, "deprecated-base") { + t.Fatalf("expected both base models, got %#v", resp) + } +} + +func TestEnrichListModelsResponse_MarksDeprecatedPricingRows(t *testing.T) { + catalog := modelCatalogForPricingJSON(t, []byte(`{ + "deprecated-model": {"provider":"openai","mode":"chat","base_model":"deprecated-model","is_deprecated":true}, + "current-model": {"provider":"openai","mode":"chat","base_model":"current-model"} + }`)) + resp := &schemas.BifrostListModelsResponse{Data: []schemas.Model{ + {ID: "openai/deprecated-model"}, + {ID: "openai/current-model"}, + {ID: "openai/provider-deprecated", IsDeprecated: true}, + }} + + enrichListModelsResponse(resp, catalog) + + if len(resp.Data) != 3 { + t.Fatalf("expected all models retained, got %#v", resp.Data) + } + byID := map[string]schemas.Model{} + for _, m := range resp.Data { + byID[m.ID] = m + } + if !byID["openai/deprecated-model"].IsDeprecated { + t.Fatalf("catalog-deprecated model should be marked deprecated: %#v", byID["openai/deprecated-model"]) + } + if byID["openai/current-model"].IsDeprecated { + t.Fatalf("current model should not be marked deprecated: %#v", byID["openai/current-model"]) + } + if !byID["openai/provider-deprecated"].IsDeprecated { + t.Fatalf("provider-deprecated flag should be preserved: %#v", byID["openai/provider-deprecated"]) + } +} + func TestListModels_UnfilteredIgnoresKeys(t *testing.T) { SetLogger(&mockLogger{}) @@ -638,7 +763,7 @@ func TestListModelDetails_SkipsDisabledKeysAndFiltersWithValid(t *testing.T) { schemas.OpenAI, []schemas.Key{ {ID: "key-a", Models: []string{"gpt-4o"}}, - {ID: "key-disabled", Enabled: boolPtr(false)}, + {ID: "key-disabled", Enabled: new(false)}, }, []string{"gpt-4o", "gpt-4o-mini"}, []string{"gpt-4o", "gpt-4o-mini"}, diff --git a/transports/bifrost-http/handlers/requestpayload_test.go b/transports/bifrost-http/handlers/requestpayload_test.go new file mode 100644 index 00000000000..64e6a7bc13e --- /dev/null +++ b/transports/bifrost-http/handlers/requestpayload_test.go @@ -0,0 +1,49 @@ +package handlers + +import ( + "strings" + "testing" + + "github.com/maximhq/bifrost/framework/configstore" + "github.com/valyala/fasthttp" +) + +type loginDecodeConfigStore struct { + configstore.ConfigStore +} + +func TestSessionLoginInvalidPayloadDoesNotExposeDecoderDetails(t *testing.T) { + h := &SessionHandler{configStore: &loginDecodeConfigStore{}} + ctx := &fasthttp.RequestCtx{} + ctx.Request.SetBodyString(`{"username":1234,"password":"Suresh"}`) + + h.login(ctx) + + body := string(ctx.Response.Body()) + if ctx.Response.StatusCode() != fasthttp.StatusBadRequest { + t.Fatalf("status = %d, want %d; body=%s", ctx.Response.StatusCode(), fasthttp.StatusBadRequest, body) + } + if !strings.Contains(body, "Invalid request payload") { + t.Fatalf("body = %s, want generic invalid payload message", body) + } + if strings.Contains(body, "cannot unmarshal") || strings.Contains(body, "username") || strings.Contains(body, "Go struct field") { + t.Fatalf("body exposes decoder internals: %s", body) + } +} + +func TestPrepareRequestInvalidPayloadDoesNotExposeDecoderDetails(t *testing.T) { + ctx := &fasthttp.RequestCtx{} + ctx.Request.SetBodyString(`{"model":1234,"prompt":"hello"}`) + + _, _, err := prepareRequest[TextRequest](ctx, nil, nil) + if err == nil { + t.Fatal("expected error for invalid payload") + } + msg := err.Error() + if msg != "Invalid request payload" { + t.Fatalf("error = %q, want generic invalid payload message", msg) + } + if strings.Contains(msg, "cannot unmarshal") || strings.Contains(msg, "model") || strings.Contains(msg, "Go struct field") { + t.Fatalf("error exposes decoder internals: %s", msg) + } +} diff --git a/transports/bifrost-http/handlers/session.go b/transports/bifrost-http/handlers/session.go index 7ac3678bb09..7c56f3e8113 100644 --- a/transports/bifrost-http/handlers/session.go +++ b/transports/bifrost-http/handlers/session.go @@ -103,7 +103,7 @@ func (h *SessionHandler) login(ctx *fasthttp.RequestCtx) { Password string `json:"password"` }{} if err := json.Unmarshal(ctx.PostBody(), &payload); err != nil { - SendError(ctx, fasthttp.StatusBadRequest, fmt.Sprintf("Invalid request format: %v", err)) + SendError(ctx, fasthttp.StatusBadRequest, "Invalid request payload") return } diff --git a/transports/bifrost-http/handlers/temp_token_scopes.go b/transports/bifrost-http/handlers/temptokens.go similarity index 80% rename from transports/bifrost-http/handlers/temp_token_scopes.go rename to transports/bifrost-http/handlers/temptokens.go index 5a5cb631321..ea7d6635873 100644 --- a/transports/bifrost-http/handlers/temp_token_scopes.go +++ b/transports/bifrost-http/handlers/temptokens.go @@ -51,6 +51,19 @@ var mcpHeadersAuthScope = temptoken.Scope{ MaxTTL: 15 * time.Minute, } +// oauth2ConsentScope authorizes the public OAuth2 consent page to call the +// consent flow APIs. The flow request ID is substituted into {id} at +// validation time, binding each token to exactly one authorize request. +var oauth2ConsentScope = temptoken.Scope{ + Name: temptoken.OAuth2ConsentScopeName, + AllowedRoutes: []temptoken.RoutePattern{ + {Method: "GET", Path: "/api/oauth2/consent/flows/{id}"}, + {Method: "PUT", Path: "/api/oauth2/consent/flows/{id}"}, + }, + ResourceIDInPath: "{id}", + MaxTTL: 15 * time.Minute, +} + // RegisterTempTokenScopes registers every scope owned by this handlers // package on the given service. Called at server startup once the service // has been constructed. Returns an error if any scope is invalid or has @@ -65,5 +78,8 @@ func RegisterTempTokenScopes(svc *temptoken.Service) error { if err := svc.Registry().Register(mcpHeadersAuthScope); err != nil { return fmt.Errorf("temp_token_scopes: register mcp_headers_auth: %w", err) } + if err := svc.Registry().Register(oauth2ConsentScope); err != nil { + return fmt.Errorf("temp_token_scopes: register oauth2_consent: %w", err) + } return nil } diff --git a/transports/bifrost-http/handlers/utils.go b/transports/bifrost-http/handlers/utils.go index c9572d9ddfc..3684bc3f3d4 100644 --- a/transports/bifrost-http/handlers/utils.go +++ b/transports/bifrost-http/handlers/utils.go @@ -5,6 +5,8 @@ package handlers import ( "encoding/json" "fmt" + "net" + "net/url" "regexp" "strings" "sync" @@ -202,13 +204,20 @@ func IsOriginAllowed(origin string, allowedOrigins []string) bool { return false } -// isLocalhostOrigin checks if the given origin is a localhost origin +// isLocalhostOrigin checks if the given origin is a localhost origin. +// Covers hostname "localhost" plus IPv4/IPv6 loopback and unspecified +// literals (127.0.0.1, ::1, 0.0.0.0, ::), bracketed or not. func isLocalhostOrigin(origin string) bool { - return strings.HasPrefix(origin, "http://localhost:") || - strings.HasPrefix(origin, "https://localhost:") || - strings.HasPrefix(origin, "http://127.0.0.1:") || - strings.HasPrefix(origin, "http://0.0.0.0:") || - strings.HasPrefix(origin, "https://127.0.0.1:") + parsed, err := url.Parse(origin) + if err != nil || (parsed.Scheme != "http" && parsed.Scheme != "https") { + return false + } + host := parsed.Hostname() // unwraps IPv6 brackets + if host == "localhost" { + return true + } + ip := net.ParseIP(host) + return ip != nil && (ip.IsLoopback() || ip.IsUnspecified()) } // wildcardRegexpCache caches compiled regexps for wildcard origin patterns. diff --git a/transports/bifrost-http/handlers/websocket.go b/transports/bifrost-http/handlers/websocket.go index 7fa84bd2a55..6f3a95bad83 100644 --- a/transports/bifrost-http/handlers/websocket.go +++ b/transports/bifrost-http/handlers/websocket.go @@ -5,6 +5,7 @@ package handlers import ( "context" "fmt" + "net" "strings" "sync" "time" @@ -69,16 +70,25 @@ func (h *WebSocketHandler) getUpgrader() websocket.FastHTTPUpgrader { // isLocalhost checks if the given host is localhost func isLocalhost(host string) bool { - // Remove port if present - if idx := strings.LastIndex(host, ":"); idx != -1 { - host = host[:idx] + // Remove port if present; SplitHostPort also unwraps IPv6 brackets ("[::1]:8080" -> "::1") + if h, _, err := net.SplitHostPort(host); err == nil { + host = h + } + // Bare bracketed IPv6 literal without a port ("[::1]"); reject half-bracketed hosts + if strings.HasPrefix(host, "[") || strings.HasSuffix(host, "]") { + if !strings.HasPrefix(host, "[") || !strings.HasSuffix(host, "]") { + return false + } + host = strings.TrimPrefix(strings.TrimSuffix(host, "]"), "[") + ip := net.ParseIP(host) + return ip != nil && ip.IsLoopback() } - // Check for localhost variations - return host == "localhost" || - host == "127.0.0.1" || - host == "::1" || - host == "" + if host == "localhost" { + return true + } + ip := net.ParseIP(host) + return ip != nil && ip.IsLoopback() } // connectStream handles WebSocket connections for real-time streaming diff --git a/transports/bifrost-http/integrations/anthropic.go b/transports/bifrost-http/integrations/anthropic.go index 624fe6278a6..dca4033ce0e 100644 --- a/transports/bifrost-http/integrations/anthropic.go +++ b/transports/bifrost-http/integrations/anthropic.go @@ -117,11 +117,20 @@ func createAnthropicMessagesRouteConfig(pathPrefix string, logger schemas.Logger ResponsesStreamResponseConverter: func(ctx *schemas.BifrostContext, resp *schemas.BifrostResponsesStreamResponse) (string, interface{}, error) { soToolName, _ := ctx.Value(schemas.BifrostContextKeyStructuredOutputToolName).(string) if soToolName == "" && shouldUsePassthrough(ctx, resp.ExtraFields.Provider, resp.ExtraFields.OriginalModelRequested, resp.ExtraFields.ResolvedModelUsed) { + anthropic.SetResponsesStreamPassthrough(ctx) // Skip passthrough for ContentPartAdded: it's a synthetic bifrost event whose // RawResponse carries the parent content_block_start already emitted by OutputItemAdded. // Passing through here would produce a duplicate content_block_start that causes // the Anthropic SDK to error and drop all subsequent content_block_delta events. - if resp.ExtraFields.RawResponse != nil && resp.Type != schemas.ResponsesStreamResponseTypeContentPartAdded { + // + // Also skip passthrough for frames the converter must renumber (see + // mustConvertInPassthrough): server tools (advisor, web_search, web_fetch, + // code_execution) expand one Responses item into several Anthropic content + // blocks, so forwarding their raw frames verbatim yields content_block indices + // the client never opened ("Content block not found" on strict clients). + if resp.ExtraFields.RawResponse != nil && + resp.Type != schemas.ResponsesStreamResponseTypeContentPartAdded && + !mustConvertInPassthrough(resp) { raw, ok := resp.ExtraFields.RawResponse.(string) if !ok { return "", nil, fmt.Errorf("expected RawResponse string, got %T", resp.ExtraFields.RawResponse) @@ -349,9 +358,76 @@ func shouldUsePassthrough(ctx *schemas.BifrostContext, provider schemas.ModelPro return anthropic.IsClaudeCodeRequest(ctx) && isClaudeModel(model, alias, string(provider)) } +// serverToolSynthesizesResultBlock reports whether an item's output_item.done +// expands into a server_tool_use content_block_stop plus a synthesized +// *_tool_result block (its own start+stop): advisor, web_search, web_fetch and +// code_execution. For these the raw upstream result-block stop cannot be +// forwarded verbatim — the client never saw a matching start — so the converter +// must render the full sequence. Computer is excluded: it streams a single +// content block and never synthesizes a result block. +func serverToolSynthesizesResultBlock(item *schemas.ResponsesMessage) bool { + if item == nil || item.Type == nil { + return false + } + switch *item.Type { + case schemas.ResponsesMessageTypeAdvisorCall, + schemas.ResponsesMessageTypeWebSearchCall, + schemas.ResponsesMessageTypeWebFetchCall, + schemas.ResponsesMessageTypeCodeInterpreterCall: + return true + } + return false +} + +// mustConvertInPassthrough reports whether a bifrost stream response must be +// rendered via the normalized converter instead of forwarding its raw upstream +// frame, even on the Anthropic passthrough path. +// +// Server tools (advisor, web_search, web_fetch, code_execution) map one Responses +// item onto several Anthropic content blocks with re-numbered indices: the forward +// converter swallows their intermediate frames (server_tool_use input delta/stop, +// result-block start) and re-synthesizes the result block at output_item.done. +// Forwarding the surviving raw frames verbatim therefore emits content_block_stop/ +// _delta for indices the client never opened, which strict clients (Claude Code) +// reject with "API Error: Content block not found". +// +// Routing the following through the converter keeps its block-index allocation +// authoritative and in lockstep with the surrounding raw frames: +// - every output_item.added runs the converter's allocator, so the block counter +// advances for each block in upstream order (server-tool starts otherwise took +// the raw path and desynced the counter); +// - server-tool output_item.done renders the server stop + synthesized result +// block start/stop at matching indices; +// - server-tool lifecycle events (no Anthropic equivalent) collapse to nothing, +// dropping the duplicate raw content_block frame they would otherwise carry. +func mustConvertInPassthrough(resp *schemas.BifrostResponsesStreamResponse) bool { + switch resp.Type { + case schemas.ResponsesStreamResponseTypeOutputItemAdded: + return true + case schemas.ResponsesStreamResponseTypeOutputItemDone: + return serverToolSynthesizesResultBlock(resp.Item) + case schemas.ResponsesStreamResponseTypeWebSearchCallInProgress, + schemas.ResponsesStreamResponseTypeWebSearchCallSearching, + schemas.ResponsesStreamResponseTypeWebSearchCallCompleted, + schemas.ResponsesStreamResponseTypeWebSearchCallResultsAdded, + schemas.ResponsesStreamResponseTypeWebSearchCallResultsCompleted, + schemas.ResponsesStreamResponseTypeWebFetchCallInProgress, + schemas.ResponsesStreamResponseTypeWebFetchCallFetching, + schemas.ResponsesStreamResponseTypeWebFetchCallCompleted, + schemas.ResponsesStreamResponseTypeCodeInterpreterCallInProgress, + schemas.ResponsesStreamResponseTypeCodeInterpreterCallInterpreting, + schemas.ResponsesStreamResponseTypeCodeInterpreterCallCompleted, + schemas.ResponsesStreamResponseTypeCodeInterpreterCallCodeDelta, + schemas.ResponsesStreamResponseTypeCodeInterpreterCallCodeDone: + return true + } + return false +} + func isClaudeModel(model, alias, provider string) bool { return (provider == string(schemas.Anthropic) || (provider == "" && (schemas.IsAnthropicModel(model) || schemas.IsAnthropicModel(alias)))) || + (provider == string(schemas.BedrockMantle) && (schemas.IsAnthropicModel(model) || schemas.IsAnthropicModel(alias))) || (provider == string(schemas.Vertex) && (schemas.IsAnthropicModel(model) || schemas.IsAnthropicModel(alias))) || (provider == string(schemas.Azure) && (schemas.IsAnthropicModel(model) || schemas.IsAnthropicModel(alias))) } @@ -861,6 +937,11 @@ func CreateAnthropicFilesRouteConfigs(pathPrefix string, handlerStore lib.Handle } uploadReq.File = fileData uploadReq.Filename = fileHeader.Filename + if contentTypeValues := form.Value["content_type"]; len(contentTypeValues) > 0 && contentTypeValues[0] != "" { + uploadReq.ContentType = &contentTypeValues[0] + } else if partContentType := strings.TrimSpace(fileHeader.Header.Get("Content-Type")); partContentType != "" { + uploadReq.ContentType = &partContentType + } return nil }, FileRequestConverter: func(ctx *schemas.BifrostContext, req any) (*FileRequest, error) { @@ -876,10 +957,11 @@ func CreateAnthropicFilesRouteConfigs(pathPrefix string, handlerStore lib.Handle return &FileRequest{ Type: schemas.FileUploadRequest, UploadRequest: &schemas.BifrostFileUploadRequest{ - File: uploadReq.File, - Filename: uploadReq.Filename, - Purpose: schemas.FilePurpose(uploadReq.Purpose), - Provider: provider, + File: uploadReq.File, + Filename: uploadReq.Filename, + Purpose: schemas.FilePurpose(uploadReq.Purpose), + ContentType: uploadReq.ContentType, + Provider: provider, }, }, nil } diff --git a/transports/bifrost-http/integrations/anthropic_test.go b/transports/bifrost-http/integrations/anthropic_test.go new file mode 100644 index 00000000000..ea76123b351 --- /dev/null +++ b/transports/bifrost-http/integrations/anthropic_test.go @@ -0,0 +1,79 @@ +package integrations + +import ( + "testing" + + "github.com/maximhq/bifrost/core/schemas" +) + +// TestMustConvertInPassthrough pins the passthrough routing decision that fixes +// the Claude Code advisor/server-tool streaming bug: server tools (advisor, +// web_search, web_fetch, code_execution) expand one Responses item into several +// Anthropic content blocks with re-numbered indices, so their frames — and every +// output_item.added (to keep the converter's block counter in lockstep) — must be +// rendered by the converter instead of forwarded raw. Computer, plain messages, +// and function/mcp tool calls stream one block each and stay on the raw path. +// +// core/providers/anthropic passthroughstream_test.go mirrors this rule for its +// end-to-end index-consistency test; keep the two in sync. +func TestMustConvertInPassthrough(t *testing.T) { + itemDone := func(mt schemas.ResponsesMessageType) *schemas.BifrostResponsesStreamResponse { + return &schemas.BifrostResponsesStreamResponse{ + Type: schemas.ResponsesStreamResponseTypeOutputItemDone, + Item: &schemas.ResponsesMessage{Type: &mt}, + } + } + typed := func(rt schemas.ResponsesStreamResponseType) *schemas.BifrostResponsesStreamResponse { + return &schemas.BifrostResponsesStreamResponse{Type: rt} + } + + cases := []struct { + name string + resp *schemas.BifrostResponsesStreamResponse + want bool + }{ + // output_item.added always converts (keeps the block counter in lockstep). + {"added_message", &schemas.BifrostResponsesStreamResponse{ + Type: schemas.ResponsesStreamResponseTypeOutputItemAdded, + Item: &schemas.ResponsesMessage{Type: schemas.Ptr(schemas.ResponsesMessageTypeMessage)}, + }, true}, + {"added_advisor", &schemas.BifrostResponsesStreamResponse{ + Type: schemas.ResponsesStreamResponseTypeOutputItemAdded, + Item: &schemas.ResponsesMessage{Type: schemas.Ptr(schemas.ResponsesMessageTypeAdvisorCall)}, + }, true}, + {"added_nil_item", typed(schemas.ResponsesStreamResponseTypeOutputItemAdded), true}, + + // output_item.done: only result-block-synthesizing server tools convert. + {"done_advisor", itemDone(schemas.ResponsesMessageTypeAdvisorCall), true}, + {"done_web_search", itemDone(schemas.ResponsesMessageTypeWebSearchCall), true}, + {"done_web_fetch", itemDone(schemas.ResponsesMessageTypeWebFetchCall), true}, + {"done_code_interpreter", itemDone(schemas.ResponsesMessageTypeCodeInterpreterCall), true}, + {"done_computer", itemDone(schemas.ResponsesMessageTypeComputerCall), false}, + {"done_message", itemDone(schemas.ResponsesMessageTypeMessage), false}, + {"done_function_call", itemDone(schemas.ResponsesMessageTypeFunctionCall), false}, + {"done_mcp_call", itemDone(schemas.ResponsesMessageTypeMCPCall), false}, + {"done_nil_item", typed(schemas.ResponsesStreamResponseTypeOutputItemDone), false}, + + // Server-tool lifecycle events convert (they collapse to nothing, dropping + // the duplicate raw content_block frame they would otherwise carry). + {"web_search_in_progress", typed(schemas.ResponsesStreamResponseTypeWebSearchCallInProgress), true}, + {"web_search_completed", typed(schemas.ResponsesStreamResponseTypeWebSearchCallCompleted), true}, + {"web_fetch_completed", typed(schemas.ResponsesStreamResponseTypeWebFetchCallCompleted), true}, + {"code_interpreter_code_done", typed(schemas.ResponsesStreamResponseTypeCodeInterpreterCallCodeDone), true}, + {"code_interpreter_completed", typed(schemas.ResponsesStreamResponseTypeCodeInterpreterCallCompleted), true}, + + // Everything else stays on the raw passthrough path. + {"text_delta", typed(schemas.ResponsesStreamResponseTypeOutputTextDelta), false}, + {"function_args_delta", typed(schemas.ResponsesStreamResponseTypeFunctionCallArgumentsDelta), false}, + {"content_part_added", typed(schemas.ResponsesStreamResponseTypeContentPartAdded), false}, + {"created", typed(schemas.ResponsesStreamResponseTypeCreated), false}, + {"completed", typed(schemas.ResponsesStreamResponseTypeCompleted), false}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if got := mustConvertInPassthrough(tc.resp); got != tc.want { + t.Errorf("mustConvertInPassthrough(%s) = %v, want %v", tc.name, got, tc.want) + } + }) + } +} diff --git a/transports/bifrost-http/integrations/bedrock_test.go b/transports/bifrost-http/integrations/bedrock_test.go index 47751ce71a3..585e599bc32 100644 --- a/transports/bifrost-http/integrations/bedrock_test.go +++ b/transports/bifrost-http/integrations/bedrock_test.go @@ -4,6 +4,8 @@ import ( "bytes" "context" "io" + "os" + "path/filepath" "strings" "testing" @@ -13,6 +15,8 @@ import ( "github.com/maximhq/bifrost/core/schemas" "github.com/maximhq/bifrost/framework/kvstore" "github.com/maximhq/bifrost/framework/logstore" + "github.com/maximhq/bifrost/framework/modelcatalog" + "github.com/maximhq/bifrost/framework/modelcatalog/datasheet" "github.com/maximhq/bifrost/transports/bifrost-http/lib" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -24,6 +28,7 @@ type mockHandlerStore struct { headerMatcher *lib.HeaderMatcher availableProviders []schemas.ModelProvider mcpHeaderCombinedAllowlist schemas.WhiteList + modelCatalog *modelcatalog.ModelCatalog } func (m *mockHandlerStore) GetHeaderMatcher() *lib.HeaderMatcher { @@ -70,9 +75,42 @@ func (m *mockHandlerStore) GetMCPExternalClientURL() string { return "" } +func (m *mockHandlerStore) GetModelCatalog() *modelcatalog.ModelCatalog { + return m.modelCatalog +} + // Ensure mockHandlerStore implements lib.HandlerStore var _ lib.HandlerStore = (*mockHandlerStore)(nil) +func TestGenericRouter_MarkDeprecatedListModelsResponseUsesCatalog(t *testing.T) { + pricingPath := filepath.Join(t.TempDir(), "pricing.json") + pricingJSON := []byte(`{ + "deprecated-model": {"provider":"openai","mode":"chat","base_model":"deprecated-model","is_deprecated":true}, + "current-model": {"provider":"openai","mode":"chat","base_model":"current-model"} + }`) + require.NoError(t, os.WriteFile(pricingPath, pricingJSON, 0o600)) + ds := datasheet.New(nil, nil, datasheet.Config{URL: "file://" + pricingPath}) + require.NoError(t, ds.LoadFromURLIntoMemory(t.Context())) + router := NewGenericRouter(nil, &mockHandlerStore{modelCatalog: modelcatalog.NewTestCatalogWithDatasheet(ds)}, nil, nil, nil) + resp := &schemas.BifrostListModelsResponse{Data: []schemas.Model{ + {ID: "openai/deprecated-model"}, + {ID: "openai/current-model"}, + {ID: "openai/provider-deprecated", IsDeprecated: true}, + }} + + router.markDeprecatedListModelsResponse(resp) + + // No models are removed; deprecated ones are flagged instead. + require.Len(t, resp.Data, 3) + byID := map[string]schemas.Model{} + for _, m := range resp.Data { + byID[m.ID] = m + } + assert.True(t, byID["openai/deprecated-model"].IsDeprecated) + assert.False(t, byID["openai/current-model"].IsDeprecated) + assert.True(t, byID["openai/provider-deprecated"].IsDeprecated) +} + func Test_parseS3URI(t *testing.T) { tests := []struct { name string @@ -407,7 +445,9 @@ func Test_createBedrockRerankRouteRequestConverter(t *testing.T) { require.NoError(t, err) require.NotNil(t, bifrostReq) require.NotNil(t, bifrostReq.RerankRequest) - assert.Equal(t, schemas.Bedrock, bifrostReq.RerankRequest.Provider) + // Converters leave Provider empty; resolution happens later in the + // modelcatalogresolver PreRequestHook. + assert.Equal(t, schemas.ModelProvider(""), bifrostReq.RerankRequest.Provider) assert.Equal(t, "capital of france", bifrostReq.RerankRequest.Query) require.Len(t, bifrostReq.RerankRequest.Documents, 1) assert.Equal(t, "Paris is capital of France", bifrostReq.RerankRequest.Documents[0].Text) diff --git a/transports/bifrost-http/integrations/genai_test.go b/transports/bifrost-http/integrations/genai_test.go index 85580117268..ea329d2b764 100644 --- a/transports/bifrost-http/integrations/genai_test.go +++ b/transports/bifrost-http/integrations/genai_test.go @@ -62,6 +62,7 @@ func TestExtractAndSetModelAndRequestTypePreservesRawBodyForGenerateContent(t *t ctx := &fasthttp.RequestCtx{} ctx.SetUserValue("model", "gemini-2.5-flash:generateContent") ctx.Request.Header.SetMethod("POST") + ctx.Request.Header.Set("x-model-provider", "gemini") ctx.Request.SetBody(rawBody) req := &gemini.GeminiGenerationRequest{} @@ -75,6 +76,27 @@ func TestExtractAndSetModelAndRequestTypePreservesRawBodyForGenerateContent(t *t assert.Equal(t, rawBody, bifrostCtx.Value(genAIRawRequestBodyContextKey)) } +func TestExtractAndSetModelAndRequestTypeNoRawPassthroughWithoutExplicitGemini(t *testing.T) { + // A bare model with no gemini/ prefix and no x-model-provider header may + // resolve to Vertex (or another provider) downstream, so the raw-body + // passthrough must not engage on the silent Gemini default. + rawBody := []byte(`{"contents":[{"role":"user","parts":[{"text":"hello"}]}]}`) + ctx := &fasthttp.RequestCtx{} + ctx.SetUserValue("model", "gemini-2.5-flash:generateContent") + ctx.Request.Header.SetMethod("POST") + ctx.Request.SetBody(rawBody) + + req := &gemini.GeminiGenerationRequest{} + require.NoError(t, sonic.Unmarshal(rawBody, req)) + bifrostCtx := schemas.NewBifrostContext(context.Background(), schemas.NoDeadline) + + err := extractAndSetModelAndRequestType(ctx, bifrostCtx, req) + require.NoError(t, err) + + assert.Nil(t, bifrostCtx.Value(schemas.BifrostContextKeyUseRawRequestBody)) + assert.Nil(t, bifrostCtx.Value(genAIRawRequestBodyContextKey)) +} + func TestExtractAndSetModelAndRequestTypeDoesNotRawPassthroughEmbedding(t *testing.T) { rawBody := []byte(`{"content":{"parts":[{"text":"hello"}]}}`) ctx := &fasthttp.RequestCtx{} @@ -98,6 +120,7 @@ func TestGenAIBatchCreateConverterCarriesRawBody(t *testing.T) { ctx := &fasthttp.RequestCtx{} ctx.SetUserValue("model", "gemini-2.5-flash:batchGenerateContent") ctx.Request.Header.SetMethod("POST") + ctx.Request.Header.Set("x-model-provider", "gemini") ctx.Request.SetBody(rawBody) req := &gemini.GeminiBatchCreateRequest{} diff --git a/transports/bifrost-http/integrations/openai.go b/transports/bifrost-http/integrations/openai.go index 50c038f5a5a..14116b5b96f 100644 --- a/transports/bifrost-http/integrations/openai.go +++ b/transports/bifrost-http/integrations/openai.go @@ -5,6 +5,7 @@ import ( "encoding/base64" "encoding/json" "errors" + "fmt" "io" "mime/multipart" "net/url" @@ -269,6 +270,19 @@ func AzureEndpointPreHook(handlerStore lib.HandlerStore) func(ctx *fasthttp.Requ } } +// openAIResponsesWireConverter maps a Bifrost responses payload to the OpenAI wire JSON shape. +func openAIResponsesWireConverter(ctx *schemas.BifrostContext, resp *schemas.BifrostResponsesResponse) (interface{}, error) { + if resp != nil && resp.ExtraFields.Provider == schemas.OpenAI { + if resp.ExtraFields.RawResponse != nil { + return resp.ExtraFields.RawResponse, nil + } + } + if resp == nil { + return nil, nil + } + return resp.WithDefaults(), nil +} + // CreateOpenAIRouteConfigs creates route configurations for OpenAI endpoints. func CreateOpenAIRouteConfigs(pathPrefix string, handlerStore lib.HandlerStore) []RouteConfig { var routes []RouteConfig @@ -704,14 +718,7 @@ func CreateOpenAIRouteConfigs(pathPrefix string, handlerStore lib.HandlerStore) } return nil, errors.New("invalid request type") }, - ResponsesResponseConverter: func(ctx *schemas.BifrostContext, resp *schemas.BifrostResponsesResponse) (interface{}, error) { - if resp.ExtraFields.Provider == schemas.OpenAI { - if resp.ExtraFields.RawResponse != nil { - return resp.ExtraFields.RawResponse, nil - } - } - return resp.WithDefaults(), nil - }, + ResponsesResponseConverter: openAIResponsesWireConverter, AsyncResponsesResponseConverter: func(ctx *schemas.BifrostContext, resp *schemas.AsyncJobResponse, responsesResponseConverter ResponsesResponseConverter) (interface{}, map[string]string, error) { bifrostResponse := &schemas.BifrostResponsesResponse{ ID: &resp.ID, @@ -800,6 +807,136 @@ func CreateOpenAIRouteConfigs(pathPrefix string, handlerStore lib.HandlerStore) }) } + // Responses lifecycle: GET retrieve + for _, path := range []string{ + "/v1/responses/{response_id}", + "/responses/{response_id}", + "/openai/responses/{response_id}", + } { + routes = append(routes, RouteConfig{ + Type: RouteConfigTypeOpenAI, + Path: pathPrefix + path, + Method: "GET", + GetHTTPRequestType: func(ctx *fasthttp.RequestCtx) schemas.RequestType { + return schemas.ResponsesRetrieveRequest + }, + GetRequestTypeInstance: func(ctx context.Context) interface{} { + return &schemas.BifrostResponsesRetrieveRequest{} + }, + PreCallback: extractResponsesLifecycleFromPath(handlerStore), + RequestConverter: func(ctx *schemas.BifrostContext, req interface{}) (*schemas.BifrostRequest, error) { + if rr, ok := req.(*schemas.BifrostResponsesRetrieveRequest); ok { + return &schemas.BifrostRequest{ + RequestType: schemas.ResponsesRetrieveRequest, + ResponsesRetrieveRequest: rr, + }, nil + } + return nil, errors.New("invalid responses retrieve request") + }, + ResponsesResponseConverter: openAIResponsesWireConverter, + ErrorConverter: func(ctx *schemas.BifrostContext, err *schemas.BifrostError) interface{} { + return err + }, + }) + } + + // Responses lifecycle: DELETE + for _, path := range []string{ + "/v1/responses/{response_id}", + "/responses/{response_id}", + "/openai/responses/{response_id}", + } { + routes = append(routes, RouteConfig{ + Type: RouteConfigTypeOpenAI, + Path: pathPrefix + path, + Method: "DELETE", + GetHTTPRequestType: func(ctx *fasthttp.RequestCtx) schemas.RequestType { + return schemas.ResponsesDeleteRequest + }, + GetRequestTypeInstance: func(ctx context.Context) interface{} { + return &schemas.BifrostResponsesDeleteRequest{} + }, + PreCallback: extractResponsesLifecycleFromPath(handlerStore), + RequestConverter: func(ctx *schemas.BifrostContext, req interface{}) (*schemas.BifrostRequest, error) { + if dr, ok := req.(*schemas.BifrostResponsesDeleteRequest); ok { + return &schemas.BifrostRequest{ + RequestType: schemas.ResponsesDeleteRequest, + ResponsesDeleteRequest: dr, + }, nil + } + return nil, errors.New("invalid responses delete request") + }, + ErrorConverter: func(ctx *schemas.BifrostContext, err *schemas.BifrostError) interface{} { + return err + }, + }) + } + + // Responses lifecycle: POST cancel + for _, path := range []string{ + "/v1/responses/{response_id}/cancel", + "/responses/{response_id}/cancel", + "/openai/responses/{response_id}/cancel", + } { + routes = append(routes, RouteConfig{ + Type: RouteConfigTypeOpenAI, + Path: pathPrefix + path, + Method: "POST", + GetHTTPRequestType: func(ctx *fasthttp.RequestCtx) schemas.RequestType { + return schemas.ResponsesCancelRequest + }, + GetRequestTypeInstance: func(ctx context.Context) interface{} { + return &schemas.BifrostResponsesCancelRequest{} + }, + PreCallback: extractResponsesLifecycleFromPath(handlerStore), + RequestConverter: func(ctx *schemas.BifrostContext, req interface{}) (*schemas.BifrostRequest, error) { + if cr, ok := req.(*schemas.BifrostResponsesCancelRequest); ok { + return &schemas.BifrostRequest{ + RequestType: schemas.ResponsesCancelRequest, + ResponsesCancelRequest: cr, + }, nil + } + return nil, errors.New("invalid responses cancel request") + }, + ResponsesResponseConverter: openAIResponsesWireConverter, + ErrorConverter: func(ctx *schemas.BifrostContext, err *schemas.BifrostError) interface{} { + return err + }, + }) + } + + // Responses lifecycle: GET input_items + for _, path := range []string{ + "/v1/responses/{response_id}/input_items", + "/responses/{response_id}/input_items", + "/openai/responses/{response_id}/input_items", + } { + routes = append(routes, RouteConfig{ + Type: RouteConfigTypeOpenAI, + Path: pathPrefix + path, + Method: "GET", + GetHTTPRequestType: func(ctx *fasthttp.RequestCtx) schemas.RequestType { + return schemas.ResponsesInputItemsRequest + }, + GetRequestTypeInstance: func(ctx context.Context) interface{} { + return &schemas.BifrostResponsesInputItemsRequest{} + }, + PreCallback: extractResponsesLifecycleFromPath(handlerStore), + RequestConverter: func(ctx *schemas.BifrostContext, req interface{}) (*schemas.BifrostRequest, error) { + if ir, ok := req.(*schemas.BifrostResponsesInputItemsRequest); ok { + return &schemas.BifrostRequest{ + RequestType: schemas.ResponsesInputItemsRequest, + ResponsesInputItemsRequest: ir, + }, nil + } + return nil, errors.New("invalid responses input items request") + }, + ErrorConverter: func(ctx *schemas.BifrostContext, err *schemas.BifrostError) interface{} { + return err + }, + }) + } + // Compaction endpoint (POST /v1/responses/compact) for _, path := range []string{ "/v1/responses/compact", @@ -2311,6 +2448,82 @@ func extractFileIDFromPath(_ lib.HandlerStore) PreRequestCallback { } } +// extractResponsesLifecycleFromPath fills response_id, provider, and query-derived fields for Responses lifecycle routes. +func extractResponsesLifecycleFromPath(_ lib.HandlerStore) PreRequestCallback { + return func(ctx *fasthttp.RequestCtx, bifrostCtx *schemas.BifrostContext, req interface{}) error { + rid := ctx.UserValue("response_id") + if rid == nil { + return errors.New("response_id is required") + } + idStr, ok := rid.(string) + if !ok || idStr == "" { + return errors.New("response_id must be a non-empty string") + } + provider := schemas.ModelProvider(string(ctx.QueryArgs().Peek("provider"))) + if provider == "" { + if isAzureSDKRequest(ctx) { + provider = schemas.Azure + } else { + provider = schemas.OpenAI + } + } + switch r := req.(type) { + case *schemas.BifrostResponsesRetrieveRequest: + r.ResponseID = idStr + r.Provider = provider + ctx.QueryArgs().VisitAll(func(key, value []byte) { + switch string(key) { + case "include": + r.Include = append(r.Include, string(value)) + } + }) + if raw := ctx.QueryArgs().Peek("starting_after"); len(raw) > 0 { + n, err := strconv.Atoi(string(raw)) + if err != nil { + return fmt.Errorf("starting_after must be an integer") + } + r.StartingAfter = schemas.Ptr(n) + } + if raw := ctx.QueryArgs().Peek("include_obfuscation"); len(raw) > 0 { + b, err := strconv.ParseBool(string(raw)) + if err != nil { + return fmt.Errorf("include_obfuscation must be a boolean") + } + r.IncludeObfuscation = &b + } + case *schemas.BifrostResponsesDeleteRequest: + r.ResponseID = idStr + r.Provider = provider + case *schemas.BifrostResponsesCancelRequest: + r.ResponseID = idStr + r.Provider = provider + case *schemas.BifrostResponsesInputItemsRequest: + r.ResponseID = idStr + r.Provider = provider + ctx.QueryArgs().VisitAll(func(key, value []byte) { + switch string(key) { + case "after": + r.After = string(value) + case "include": + r.Include = append(r.Include, string(value)) + case "order": + r.Order = string(value) + } + }) + if raw := ctx.QueryArgs().Peek("limit"); len(raw) > 0 { + n, err := strconv.Atoi(string(raw)) + if err != nil { + return fmt.Errorf("limit must be an integer") + } + r.Limit = schemas.Ptr(n) + } + default: + return errors.New("invalid request type for responses lifecycle path") + } + return nil + } +} + // parseOpenAIFileUploadMultipartRequest parses multipart/form-data for file upload requests func parseOpenAIFileUploadMultipartRequest(ctx *fasthttp.RequestCtx, req interface{}) error { uploadReq, ok := req.(*schemas.BifrostFileUploadRequest) @@ -2353,6 +2566,12 @@ func parseOpenAIFileUploadMultipartRequest(ctx *fasthttp.RequestCtx, req interfa uploadReq.File = fileData uploadReq.Filename = fileHeader.Filename + if contentTypeValues := form.Value["content_type"]; len(contentTypeValues) > 0 && contentTypeValues[0] != "" { + uploadReq.ContentType = &contentTypeValues[0] + } else if partContentType := strings.TrimSpace(fileHeader.Header.Get("Content-Type")); partContentType != "" { + uploadReq.ContentType = &partContentType + } + // Extract provider from extra_body (form field) if providerValues := form.Value["provider"]; len(providerValues) > 0 && providerValues[0] != "" { uploadReq.Provider = schemas.ModelProvider(providerValues[0]) @@ -3111,6 +3330,22 @@ func parseTranscriptionMultipartRequest(ctx *fasthttp.RequestCtx, req interface{ transcriptionReq.Stream = &stream } + // chunking_strategy is OpenAI-specific (required by diarization models). It is a + // Union["auto", server_vad object], so decode object-shaped values and pass + // plain strings (e.g. "auto") through verbatim via ExtraParams passthrough. + if csValues := form.Value["chunking_strategy"]; len(csValues) > 0 && csValues[0] != "" { + raw := csValues[0] + if transcriptionReq.ExtraParams == nil { + transcriptionReq.ExtraParams = map[string]interface{}{} + } + var obj map[string]interface{} + if err := json.Unmarshal([]byte(raw), &obj); err == nil { + transcriptionReq.ExtraParams["chunking_strategy"] = obj + } else { + transcriptionReq.ExtraParams["chunking_strategy"] = raw + } + } + return nil } @@ -3545,4 +3780,4 @@ func parseContainerFileCreateMultipartRequest(ctx *fasthttp.RequestCtx, req inte } return nil -} +} \ No newline at end of file diff --git a/transports/bifrost-http/integrations/router.go b/transports/bifrost-http/integrations/router.go index d916f254e16..640c262e290 100644 --- a/transports/bifrost-http/integrations/router.go +++ b/transports/bifrost-http/integrations/router.go @@ -65,6 +65,7 @@ import ( "github.com/maximhq/bifrost/core/providers/bedrock" "github.com/maximhq/bifrost/core/schemas" "github.com/maximhq/bifrost/framework/logstore" + "github.com/maximhq/bifrost/framework/modelcatalog" "github.com/maximhq/bifrost/transports/bifrost-http/lib" "github.com/valyala/fasthttp" ) @@ -544,6 +545,10 @@ type GenericRouter struct { largeResponseHook LargeResponseHook // Optional: enterprise hook for large response scanning } +type modelCatalogProvider interface { + GetModelCatalog() *modelcatalog.ModelCatalog +} + // SetLargePayloadHook sets the hook for large payload detection and streaming. // This is used by enterprise to inject large payload optimization without // embedding the logic in the OSS router. @@ -948,6 +953,7 @@ func (g *GenericRouter) handleNonStreamingRequest(ctx *fasthttp.RequestCtx, conf g.sendError(ctx, bifrostCtx, config.ErrorConverter, newBifrostError(nil, "Bifrost response is nil after post-request callback")) return } + g.markDeprecatedListModelsResponse(listModelsResponse) response, err = config.ListModelsResponseConverter(bifrostCtx, listModelsResponse) bifrostExtraFields = listModelsResponse.ExtraFields @@ -1448,6 +1454,90 @@ func (g *GenericRouter) handleNonStreamingRequest(ctx *fasthttp.RequestCtx, conf response, err = config.VideoListResponseConverter(bifrostCtx, videoListResponse) bifrostExtraFields = videoListResponse.ExtraFields + case bifrostReq.ResponsesRetrieveRequest != nil: + responsesRetrieveResponse, bifrostErr := g.client.ResponsesRetrieveRequest(bifrostCtx, bifrostReq.ResponsesRetrieveRequest) + if bifrostErr != nil { + g.sendError(ctx, bifrostCtx, config.ErrorConverter, bifrostErr) + return + } + if config.PostCallback != nil { + if err := config.PostCallback(ctx, req, responsesRetrieveResponse); err != nil { + g.sendError(ctx, bifrostCtx, config.ErrorConverter, newBifrostError(err, "failed to execute post-request callback")) + return + } + } + if responsesRetrieveResponse == nil { + g.sendError(ctx, bifrostCtx, config.ErrorConverter, newBifrostError(nil, "Bifrost response is nil after post-request callback")) + return + } + if config.ResponsesResponseConverter == nil { + g.sendError(ctx, bifrostCtx, config.ErrorConverter, newBifrostError(nil, "missing ResponsesResponseConverter for integration")) + return + } + response, err = config.ResponsesResponseConverter(bifrostCtx, responsesRetrieveResponse) + bifrostExtraFields = responsesRetrieveResponse.ExtraFields + + case bifrostReq.ResponsesDeleteRequest != nil: + responsesDeleteResponse, bifrostErr := g.client.ResponsesDeleteRequest(bifrostCtx, bifrostReq.ResponsesDeleteRequest) + if bifrostErr != nil { + g.sendError(ctx, bifrostCtx, config.ErrorConverter, bifrostErr) + return + } + if config.PostCallback != nil { + if err := config.PostCallback(ctx, req, responsesDeleteResponse); err != nil { + g.sendError(ctx, bifrostCtx, config.ErrorConverter, newBifrostError(err, "failed to execute post-request callback")) + return + } + } + if responsesDeleteResponse == nil { + g.sendError(ctx, bifrostCtx, config.ErrorConverter, newBifrostError(nil, "Bifrost response is nil after post-request callback")) + return + } + response = responsesDeleteResponse + bifrostExtraFields = responsesDeleteResponse.ExtraFields + + case bifrostReq.ResponsesCancelRequest != nil: + responsesCancelResponse, bifrostErr := g.client.ResponsesCancelRequest(bifrostCtx, bifrostReq.ResponsesCancelRequest) + if bifrostErr != nil { + g.sendError(ctx, bifrostCtx, config.ErrorConverter, bifrostErr) + return + } + if config.PostCallback != nil { + if err := config.PostCallback(ctx, req, responsesCancelResponse); err != nil { + g.sendError(ctx, bifrostCtx, config.ErrorConverter, newBifrostError(err, "failed to execute post-request callback")) + return + } + } + if responsesCancelResponse == nil { + g.sendError(ctx, bifrostCtx, config.ErrorConverter, newBifrostError(nil, "Bifrost response is nil after post-request callback")) + return + } + if config.ResponsesResponseConverter == nil { + g.sendError(ctx, bifrostCtx, config.ErrorConverter, newBifrostError(nil, "missing ResponsesResponseConverter for integration")) + return + } + response, err = config.ResponsesResponseConverter(bifrostCtx, responsesCancelResponse) + bifrostExtraFields = responsesCancelResponse.ExtraFields + + case bifrostReq.ResponsesInputItemsRequest != nil: + inputItemsResponse, bifrostErr := g.client.ResponsesInputItemsRequest(bifrostCtx, bifrostReq.ResponsesInputItemsRequest) + if bifrostErr != nil { + g.sendError(ctx, bifrostCtx, config.ErrorConverter, bifrostErr) + return + } + if config.PostCallback != nil { + if err := config.PostCallback(ctx, req, inputItemsResponse); err != nil { + g.sendError(ctx, bifrostCtx, config.ErrorConverter, newBifrostError(err, "failed to execute post-request callback")) + return + } + } + if inputItemsResponse == nil { + g.sendError(ctx, bifrostCtx, config.ErrorConverter, newBifrostError(nil, "Bifrost response is nil after post-request callback")) + return + } + response = inputItemsResponse + bifrostExtraFields = inputItemsResponse.ExtraFields + case bifrostReq.CountTokensRequest != nil: countTokensResponse, bifrostErr := g.client.CountTokensRequest(bifrostCtx, bifrostReq.CountTokensRequest) if bifrostErr != nil { @@ -1692,6 +1782,32 @@ func (g *GenericRouter) handleAsyncJobResponse(ctx *fasthttp.RequestCtx, bifrost } } +// markDeprecatedListModelsResponse annotates deprecated models with the +// IsDeprecated flag using catalog pricing data, without removing them from the +// response. Clients decide how to surface deprecated entries (e.g. the UI keeps +// them visible but non-selectable). +func (g *GenericRouter) markDeprecatedListModelsResponse(resp *schemas.BifrostListModelsResponse) { + if resp == nil || len(resp.Data) == 0 { + return + } + catalogProvider, ok := g.handlerStore.(modelCatalogProvider) + if !ok || catalogProvider.GetModelCatalog() == nil { + return + } + catalog := catalogProvider.GetModelCatalog() + for i := range resp.Data { + model := resp.Data[i] + provider, modelName := schemas.ParseModelString(model.ID, "") + pricingEntry := catalog.GetPricingEntryForModel(modelName, provider) + if pricingEntry == nil && model.Alias != nil { + pricingEntry = catalog.GetPricingEntryForModel(*model.Alias, provider) + } + if pricingEntry != nil && pricingEntry.IsDeprecated { + resp.Data[i].IsDeprecated = true + } + } +} + // handleBatchRequest handles batch API requests (create, list, retrieve, cancel, results) func (g *GenericRouter) handleBatchRequest(ctx *fasthttp.RequestCtx, config RouteConfig, req interface{}, batchReq *BatchRequest, bifrostCtx *schemas.BifrostContext) { var response interface{} diff --git a/transports/bifrost-http/integrations/router_test.go b/transports/bifrost-http/integrations/router_test.go index 7acbc153188..35594e8a349 100644 --- a/transports/bifrost-http/integrations/router_test.go +++ b/transports/bifrost-http/integrations/router_test.go @@ -377,10 +377,15 @@ func TestOpenAIChatStructuredOutputRequestParserAndConverter(t *testing.T) { assert.Contains(t, responseFormat, "json_schema") } -func TestCreateHandler_AnthropicRouteClears_UseRawRequestBody_WhenCatalogSelectsBedrock(t *testing.T) { - handlerStore := &mockHandlerStore{ - availableProviders: []schemas.ModelProvider{schemas.Bedrock}, - } +// TestCreateHandler_AnthropicRouteSetsPassthroughFlags verifies that a Claude +// Code request on the Anthropic route is marked for raw-body passthrough by the +// checkAnthropicPassthrough pre-callback, and that the flags are still set at +// converter time. The router does not clear them when the model later resolves +// to a non-native provider (e.g. Bedrock) — that happens per attempt in core +// (clearAnthropicPassthroughForNonNativeProvider), after final provider +// resolution. +func TestCreateHandler_AnthropicRouteSetsPassthroughFlags(t *testing.T) { + handlerStore := &mockHandlerStore{} var capturedUseRaw interface{} var capturedSendRawResponse interface{} @@ -415,10 +420,12 @@ func TestCreateHandler_AnthropicRouteClears_UseRawRequestBody_WhenCatalogSelects router.createHandler(route)(ctx) - require.Equal(t, fasthttp.StatusInternalServerError, ctx.Response.StatusCode()) - require.Equal(t, false, capturedUseRaw, "UseRawRequestBody should be cleared when catalog selects Bedrock") - require.Equal(t, false, capturedSendRawResponse, "SendBackRawResponse should be cleared when catalog selects Bedrock") - require.Equal(t, false, capturedPassthroughOverrides, "PassthroughOverridesPresent should be cleared when catalog selects Bedrock") + // Non-Bifrost errors without an explicit status code map to 400 (see + // GenericRouter.sendError), so the converter's sentinel error surfaces as one. + require.Equal(t, fasthttp.StatusBadRequest, ctx.Response.StatusCode()) + require.Equal(t, true, capturedUseRaw, "UseRawRequestBody should be set for a Claude Code request") + require.Equal(t, true, capturedSendRawResponse, "SendBackRawResponse should be set for a Claude Code request") + require.Equal(t, true, capturedPassthroughOverrides, "PassthroughOverridesPresent should be set for a Claude Code request") } func TestCreateHandler_CustomParserFailureClosesConnection(t *testing.T) { diff --git a/transports/bifrost-http/lib/config.go b/transports/bifrost-http/lib/config.go index 1775140ba70..61bc0cf061c 100644 --- a/transports/bifrost-http/lib/config.go +++ b/transports/bifrost-http/lib/config.go @@ -491,6 +491,15 @@ type Config struct { LogsStoreConfig *logstore.Config ObjectStore objectstore.ObjectStore + // oauth2SigningKey caches the immutable OAuth2 signing key used to sign and + // verify Bifrost-issued /mcp JWTs. The key is created once via an idempotent + // insert and never rotated, so it is identical across nodes and immutable for + // the process lifetime. Caching it here lets the JWKS, token-issuance, and + // JWT-verify paths share a single load — sparing each a DB read + private-key + // decrypt per request — through one invalidation point. See + // GetOAuth2SigningKey. + oauth2SigningKey atomic.Pointer[configstoreTables.OAuth2SigningKey] + // In-memory storage ServerConfig *ServerConfig ClientConfig *configstore.ClientConfig @@ -868,6 +877,12 @@ func LoadConfig(ctx context.Context, configDirPath string) (*Config, error) { } // 4. Client config (store → file → defaults) loadClientConfig(ctx, config, &configData) + // Reject an out-of-range client config (e.g. auth_code_ttl above the cap) + // loudly at startup instead of silently correcting it, so in-memory, core, + // and DB state cannot diverge. + if err := validateClientConfig(config.ClientConfig); err != nil { + return nil, err + } config.SetHeaderMatcher(NewHeaderMatcher(config.ClientConfig.HeaderFilterConfig)) // 5. Providers (store → file → auto-detect) if err := loadProviders(ctx, config, &configData); err != nil { @@ -1077,6 +1092,23 @@ func applyClientConfigDefaults(cc *configstore.ClientConfig) { } } +// validateClientConfig checks invariants on a fully-merged client config that +// must hold regardless of the source (config.json, DB, or defaults). It returns +// an error rather than silently correcting a value so an out-of-range setting +// fails loudly at startup instead of diverging in-memory state from what the +// operator wrote — and stays wrong (re-warned, unpersisted) on every restart. +func validateClientConfig(cc *configstore.ClientConfig) error { + // The /api/config handler rejects an over-max auth_code_ttl, but config.json + // and any DB row written before the cap existed bypass that path. Fail fast + // here, the point where every config source converges, so a leaked one-time + // code can never be minted with a lifetime above the ceiling. A zero/omitted + // value is valid — it resolves to the default at issuance. + if oc := cc.OAuth2ServerConfig; oc != nil && oc.AuthCodeTTL > configstoreTables.MaxAuthCodeTTL { + return fmt.Errorf("oauth2_server_config.auth_code_ttl %d exceeds the maximum of %d seconds (15 minutes)", oc.AuthCodeTTL, configstoreTables.MaxAuthCodeTTL) + } + return nil +} + // sanitizeMCPExternalOAuthURLs validates the MCP external OAuth URL overrides // on a ClientConfig and clears any invalid override so it cannot leak into // OAuth URL generation. The warning intentionally omits the offending value: @@ -1450,21 +1482,22 @@ func mergeProviderKeys(provider schemas.ModelProvider, fileKeys, dbKeys []schema } else { // No stored hash (legacy) - fall back to generating fresh hash dbKeyHash, err := configstore.GenerateKeyHash(schemas.Key{ - Name: dbKey.Name, - Value: dbKey.Value, - Models: dbKey.Models, - BlacklistedModels: dbKey.BlacklistedModels, - Weight: dbKey.Weight, - AzureKeyConfig: dbKey.AzureKeyConfig, - VertexKeyConfig: dbKey.VertexKeyConfig, - BedrockKeyConfig: dbKey.BedrockKeyConfig, - ReplicateKeyConfig: dbKey.ReplicateKeyConfig, - Aliases: dbKey.Aliases, - VLLMKeyConfig: dbKey.VLLMKeyConfig, - OllamaKeyConfig: dbKey.OllamaKeyConfig, - SGLKeyConfig: dbKey.SGLKeyConfig, - Enabled: dbKey.Enabled, - UseForBatchAPI: dbKey.UseForBatchAPI, + Name: dbKey.Name, + Value: dbKey.Value, + Models: dbKey.Models, + BlacklistedModels: dbKey.BlacklistedModels, + Weight: dbKey.Weight, + AzureKeyConfig: dbKey.AzureKeyConfig, + VertexKeyConfig: dbKey.VertexKeyConfig, + BedrockKeyConfig: dbKey.BedrockKeyConfig, + BedrockMantleKeyConfig: dbKey.BedrockMantleKeyConfig, + ReplicateKeyConfig: dbKey.ReplicateKeyConfig, + Aliases: dbKey.Aliases, + VLLMKeyConfig: dbKey.VLLMKeyConfig, + OllamaKeyConfig: dbKey.OllamaKeyConfig, + SGLKeyConfig: dbKey.SGLKeyConfig, + Enabled: dbKey.Enabled, + UseForBatchAPI: dbKey.UseForBatchAPI, }) if err != nil { logger.Warn("failed to generate key hash for db key %s (%s): %v, falling back to name comparison", dbKey.Name, provider, err) @@ -1531,21 +1564,22 @@ func reconcileProviderKeys(provider schemas.ModelProvider, fileKeys, dbKeys []sc } else { // No stored hash (legacy) - fall back to generating fresh hash for comparison dbKeyHash, err := configstore.GenerateKeyHash(schemas.Key{ - Name: dbKey.Name, - Value: dbKey.Value, - Models: dbKey.Models, - BlacklistedModels: dbKey.BlacklistedModels, - Weight: dbKey.Weight, - AzureKeyConfig: dbKey.AzureKeyConfig, - VertexKeyConfig: dbKey.VertexKeyConfig, - BedrockKeyConfig: dbKey.BedrockKeyConfig, - ReplicateKeyConfig: dbKey.ReplicateKeyConfig, - Aliases: dbKey.Aliases, - VLLMKeyConfig: dbKey.VLLMKeyConfig, - OllamaKeyConfig: dbKey.OllamaKeyConfig, - SGLKeyConfig: dbKey.SGLKeyConfig, - Enabled: dbKey.Enabled, - UseForBatchAPI: dbKey.UseForBatchAPI, + Name: dbKey.Name, + Value: dbKey.Value, + Models: dbKey.Models, + BlacklistedModels: dbKey.BlacklistedModels, + Weight: dbKey.Weight, + AzureKeyConfig: dbKey.AzureKeyConfig, + VertexKeyConfig: dbKey.VertexKeyConfig, + BedrockKeyConfig: dbKey.BedrockKeyConfig, + BedrockMantleKeyConfig: dbKey.BedrockMantleKeyConfig, + ReplicateKeyConfig: dbKey.ReplicateKeyConfig, + Aliases: dbKey.Aliases, + VLLMKeyConfig: dbKey.VLLMKeyConfig, + OllamaKeyConfig: dbKey.OllamaKeyConfig, + SGLKeyConfig: dbKey.SGLKeyConfig, + Enabled: dbKey.Enabled, + UseForBatchAPI: dbKey.UseForBatchAPI, }) if err != nil { logger.Warn("failed to generate key hash for db key %s (%s): %v", dbKey.Name, provider, err) @@ -1789,6 +1823,12 @@ func mcpClientConfigToTable(clientConfig *schemas.MCPClientConfig) (configstoreT clientConfig.ToolSyncInterval.String(), ) } + if clientConfig.ToolExecutionTimeout < 0 { + return configstoreTables.TableMCPClient{}, fmt.Errorf( + "tool_execution_timeout must be >= 0, got %q", + clientConfig.ToolExecutionTimeout.String(), + ) + } authType := string(clientConfig.AuthType) if authType == "" { authType = string(schemas.MCPAuthTypeHeaders) @@ -1808,6 +1848,7 @@ func mcpClientConfigToTable(clientConfig *schemas.MCPClientConfig) (configstoreT AllowedExtraHeaders: clientConfig.AllowedExtraHeaders, IsPingAvailable: clientConfig.IsPingAvailable, ToolSyncInterval: int(clientConfig.ToolSyncInterval / time.Second), + ToolExecutionTimeout: int(math.Ceil(clientConfig.ToolExecutionTimeout.Seconds())), ToolPricing: clientConfig.ToolPricing, AllowOnAllVirtualKeys: clientConfig.AllowOnAllVirtualKeys, Disabled: clientConfig.Disabled, @@ -2251,26 +2292,18 @@ func mergeGovernanceConfig(ctx context.Context, config *Config, configData *Conf if forceFileSync || existingVirtualKey.ConfigHash != fileVKHash { logger.Debug("config hash mismatch for virtual key %s, syncing from config file", existingVirtualKey.ID) configData.Governance.VirtualKeys[i].ConfigHash = fileVKHash - // This is added for backward compatibility with existing configs - if configData.Governance.VirtualKeys[i].Value == "" && existingVirtualKey.Value != "" { + // Preserve stored value when config doesn't supply one + if configData.Governance.VirtualKeys[i].Value.ShouldPreserveStored() && existingVirtualKey.Value.IsSet() { configData.Governance.VirtualKeys[i].Value = existingVirtualKey.Value } - // Process environment variable for virtual key value - if strings.HasPrefix(configData.Governance.VirtualKeys[i].Value, "env.") { - // Resolving the environment variable value - envValue, err := envutils.ProcessEnvValue(configData.Governance.VirtualKeys[i].Value) - if err != nil { - logger.Warn("failed to process environment variable for virtual key %s: %v", newVirtualKey.ID, err) - continue - } - configData.Governance.VirtualKeys[i].Value = envValue + resolvedVal := configData.Governance.VirtualKeys[i].Value.GetValue() + if resolvedVal == "" && configData.Governance.VirtualKeys[i].Value.IsFromSecret() { + logger.Warn("virtual key %s: env/vault ref %q could not be resolved, skipping update", newVirtualKey.ID, configData.Governance.VirtualKeys[i].Value.GetRawRef()) + break } - // If the virtual key value is not a valid virtual key, we will generate a new one - if !strings.HasPrefix(configData.Governance.VirtualKeys[i].Value, governance.VirtualKeyPrefix) { - if configData.Governance.VirtualKeys[i].Value != "" { - logger.Warn("virtual key %s has a value in the config file that does not have %s prefix. We are generating a new one for you.", newVirtualKey.ID, governance.VirtualKeyPrefix) - } - configData.Governance.VirtualKeys[i].Value = governance.GenerateVirtualKey() + if !strings.HasPrefix(resolvedVal, governance.VirtualKeyPrefix) { + logger.Warn("virtual key %s has a value in the config file that does not have %s prefix. We are generating a new one for you.", newVirtualKey.ID, governance.VirtualKeyPrefix) + configData.Governance.VirtualKeys[i].Value = *schemas.NewSecretVar(governance.GenerateVirtualKey()) } // Resolve MCP client names to IDs for config file mcp_configs configData.Governance.VirtualKeys[i].MCPConfigs = resolveMCPConfigClientIDs( @@ -2288,22 +2321,14 @@ func mergeGovernanceConfig(ctx context.Context, config *Config, configData *Conf if configData.Governance.VirtualKeys[i].ID == "" { configData.Governance.VirtualKeys[i].ID = uuid.NewString() } - // if the virtual key value is env.VIRTUAL_KEY_VALUE, then we will need to resolve the environment variable - // Process environment variable for virtual key value - if strings.HasPrefix(configData.Governance.VirtualKeys[i].Value, "env.") { - // Resolving the environment variable value - envValue, err := envutils.ProcessEnvValue(configData.Governance.VirtualKeys[i].Value) - if err != nil { - logger.Warn("failed to process environment variable for virtual key %s: %v", newVirtualKey.ID, err) - continue - } - configData.Governance.VirtualKeys[i].Value = envValue + resolvedVal := configData.Governance.VirtualKeys[i].Value.GetValue() + if resolvedVal == "" && configData.Governance.VirtualKeys[i].Value.IsFromSecret() { + logger.Warn("virtual key %s: env/vault ref %q could not be resolved, skipping", newVirtualKey.ID, configData.Governance.VirtualKeys[i].Value.GetRawRef()) + continue } - if !strings.HasPrefix(configData.Governance.VirtualKeys[i].Value, governance.VirtualKeyPrefix) { - if configData.Governance.VirtualKeys[i].Value != "" { - logger.Warn("virtual key %s has a value in the config file that does not have %s prefix. We are generating a new one for you.", newVirtualKey.ID, governance.VirtualKeyPrefix) - } - configData.Governance.VirtualKeys[i].Value = governance.GenerateVirtualKey() + if !strings.HasPrefix(resolvedVal, governance.VirtualKeyPrefix) { + logger.Warn("virtual key %s has a value in the config file that does not have %s prefix. We are generating a new one for you.", newVirtualKey.ID, governance.VirtualKeyPrefix) + configData.Governance.VirtualKeys[i].Value = *schemas.NewSecretVar(governance.GenerateVirtualKey()) } // Resolve MCP client names to IDs for config file mcp_configs configData.Governance.VirtualKeys[i].MCPConfigs = resolveMCPConfigClientIDs( @@ -3883,7 +3908,7 @@ func ResolveFrameworkPricingConfig( // Hash the file-resolved values; skip if nothing valid survived Phase 1. fileHash := "" fileHasHashableMCPConfig := (fileMCPLibraryURL != nil && !skipMCPLibraryURLBackfill) || fileMCPLibrarySyncSeconds != nil - if fileConfig != nil && fileConfig.Pricing != nil && !skipURLBackfill && (filePricingURL != nil || fileSyncSeconds != nil || fileHasHashableMCPConfig) { + if fileConfig != nil && fileConfig.Pricing != nil && !skipURLBackfill && (filePricingURL != nil || (fileModelParametersURL != nil && !skipModelParamsURLBackfill) || fileSyncSeconds != nil || fileHasHashableMCPConfig) { var h string var err error if fileHasHashableMCPConfig { @@ -3925,7 +3950,12 @@ func ResolveFrameworkPricingConfig( needsDBUpdate = true } if dbConfig.ModelParametersURL != nil && *dbConfig.ModelParametersURL != "" { - resolvedModelParametersURL = dbConfig.ModelParametersURL + if fileChanged && fileModelParametersURL != nil && !skipModelParamsURLBackfill { + logger.Info("model_parameters_url from config.json overrides DB (file hash changed) — updating DB") + needsDBUpdate = true + } else { + resolvedModelParametersURL = dbConfig.ModelParametersURL + } } else if !skipModelParamsURLBackfill { needsDBUpdate = true } @@ -4330,29 +4360,6 @@ func (c *Config) GetRawConfigString() string { return string(data) } -// processEnvValue checks and replaces environment variable references in configuration values. -// Returns the processed value and the environment variable name if it was an env reference. -// Supports the "env.VARIABLE_NAME" syntax for referencing environment variables. -// This enables secure configuration management without hardcoding sensitive values. -// -// Examples: -// - "env.OPENAI_API_KEY" -> actual value from OPENAI_API_KEY environment variable -// - "sk-1234567890" -> returned as-is (no env prefix) -func (c *Config) processEnvValue(value string) (string, string, error) { - v := strings.TrimSpace(value) - if !strings.HasPrefix(v, "env.") { - return value, "", nil // do not trim non-env values - } - envKey := strings.TrimSpace(strings.TrimPrefix(v, "env.")) - if envKey == "" { - return "", "", fmt.Errorf("environment variable name missing in %q", value) - } - if envValue, ok := os.LookupEnv(envKey); ok { - return envValue, envKey, nil - } - return "", envKey, fmt.Errorf("environment variable %s not found", envKey) -} - // GetProviderConfigRaw retrieves the raw, unredacted provider configuration from memory. // This method is for internal use only, particularly by the account implementation. // @@ -4404,6 +4411,13 @@ func (c *Config) GetHeaderMatcher() *HeaderMatcher { return c.headerMatcher.Load() } +func (c *Config) GetModelCatalog() *modelcatalog.ModelCatalog { + if c == nil { + return nil + } + return c.ModelCatalog +} + // SetHeaderMatcher atomically stores a new precompiled header matcher. // Called when header filter config changes. func (c *Config) SetHeaderMatcher(m *HeaderMatcher) { @@ -5593,6 +5607,17 @@ func (c *Config) GetAllKeys() ([]configstoreTables.TableKey, error) { cfg.SessionToken = cfg.SessionToken.Redacted() configStoreKey.BedrockKeyConfig = &cfg } + if key.BedrockMantleKeyConfig != nil { + cfg := *key.BedrockMantleKeyConfig // safe copy + cfg.AccessKey = *cfg.AccessKey.Redacted() + cfg.SecretKey = *cfg.SecretKey.Redacted() + cfg.SessionToken = cfg.SessionToken.Redacted() + cfg.Region = cfg.Region.Redacted() + cfg.RoleARN = cfg.RoleARN.Redacted() + cfg.ExternalID = cfg.ExternalID.Redacted() + cfg.RoleSessionName = cfg.RoleSessionName.Redacted() + configStoreKey.BedrockMantleKeyConfig = &cfg + } if key.VertexKeyConfig != nil { cfg := *key.VertexKeyConfig // safe copy cfg.ProjectID = *cfg.ProjectID.Redacted() @@ -5786,6 +5811,7 @@ func (c *Config) UpdateMCPClient(ctx context.Context, id string, updatedConfig * c.MCPConfig.ClientConfigs[configIndex].ToolPricing = updatedConfig.ToolPricing c.MCPConfig.ClientConfigs[configIndex].IsPingAvailable = updatedConfig.IsPingAvailable c.MCPConfig.ClientConfigs[configIndex].ToolSyncInterval = updatedConfig.ToolSyncInterval + c.MCPConfig.ClientConfigs[configIndex].ToolExecutionTimeout = updatedConfig.ToolExecutionTimeout c.MCPConfig.ClientConfigs[configIndex].AllowOnAllVirtualKeys = updatedConfig.AllowOnAllVirtualKeys c.MCPConfig.ClientConfigs[configIndex].Disabled = updatedConfig.Disabled c.MCPConfig.ClientConfigs[configIndex].PerUserHeaderKeys = updatedConfig.PerUserHeaderKeys @@ -6021,6 +6047,27 @@ func (c *Config) RedactMCPClientConfig(config *schemas.MCPClientConfig) *schemas return &configCopy } +// GetOAuth2SigningKey returns the OAuth2 signing key, loading it from the +// config store on first use and caching it for the process lifetime. The key is +// immutable once created (see the oauth2SigningKey field), so a process-lifetime +// cache is safe and lets the JWKS, token-issuance, and JWT-verify paths share a +// single load — skipping a DB read + private-key decrypt per request. Returns an +// error when no config store is wired. +func (c *Config) GetOAuth2SigningKey(ctx context.Context) (*configstoreTables.OAuth2SigningKey, error) { + if k := c.oauth2SigningKey.Load(); k != nil { + return k, nil + } + if c.ConfigStore == nil { + return nil, fmt.Errorf("config store unavailable") + } + k, err := c.ConfigStore.GetOAuth2SigningKey(ctx) + if err != nil { + return nil, err + } + c.oauth2SigningKey.Store(k) + return k, nil +} + // autoDetectProviders automatically detects common environment variables and sets up providers // when no configuration file exists. This enables zero-config startup when users have set // standard environment variables like OPENAI_API_KEY, ANTHROPIC_API_KEY, etc. @@ -6046,7 +6093,7 @@ func (c *Config) autoDetectProviders(ctx context.Context) { for provider, envVars := range providerEnvVars { for _, envVar := range envVars { - if apiKey := os.Getenv(envVar); apiKey != "" { + if os.Getenv(envVar) != "" { // Generate a unique ID for the auto-detected key keyID := uuid.NewString() // Create default provider configuration @@ -6055,7 +6102,7 @@ func (c *Config) autoDetectProviders(ctx context.Context) { { ID: keyID, Name: fmt.Sprintf("%s_auto_detected", envVar), - Value: *schemas.NewSecretVar(apiKey), + Value: *schemas.NewSecretVar("env." + envVar), Models: schemas.WhiteList{"*"}, Weight: 1.0, }, diff --git a/transports/bifrost-http/lib/config_test.go b/transports/bifrost-http/lib/config_test.go index 12d58d42e04..1e23526436f 100644 --- a/transports/bifrost-http/lib/config_test.go +++ b/transports/bifrost-http/lib/config_test.go @@ -425,6 +425,63 @@ func NewMockConfigStore() *MockConfigStore { func (m *MockConfigStore) RefreshConnectionPool(ctx context.Context) error { return nil } +func (m *MockConfigStore) GetOAuth2SigningKey(ctx context.Context) (*tables.OAuth2SigningKey, error) { + return &tables.OAuth2SigningKey{}, nil +} +func (m *MockConfigStore) CreateOAuth2Client(ctx context.Context, client *tables.TableOAuth2Client) error { + return nil +} +func (m *MockConfigStore) GetOAuth2ClientByClientID(ctx context.Context, clientID string) (*tables.TableOAuth2Client, error) { + return nil, configstore.ErrNotFound +} +func (m *MockConfigStore) CreateOAuth2AuthorizeRequest(ctx context.Context, req *tables.TableOAuth2AuthorizeRequest) error { + return nil +} +func (m *MockConfigStore) GetOAuth2AuthorizeRequestByID(ctx context.Context, id string) (*tables.TableOAuth2AuthorizeRequest, error) { + return nil, configstore.ErrNotFound +} +func (m *MockConfigStore) GetOAuth2AuthorizeRequestByCodeHash(ctx context.Context, codeHash string) (*tables.TableOAuth2AuthorizeRequest, error) { + return nil, configstore.ErrNotFound +} +func (m *MockConfigStore) ConsentOAuth2AuthorizeRequest(ctx context.Context, req *tables.TableOAuth2AuthorizeRequest) error { + return nil +} +func (m *MockConfigStore) SweepExpiredOAuth2AuthorizeRequests(ctx context.Context) error { + return nil +} +func (m *MockConfigStore) GetOAuth2RefreshTokenByHash(ctx context.Context, hash string) (*tables.TableOAuth2RefreshToken, error) { + return nil, configstore.ErrNotFound +} +func (m *MockConfigStore) ConsumeOAuth2AuthorizeRequest(ctx context.Context, requestID string, rt *tables.TableOAuth2RefreshToken) error { + return nil +} +func (m *MockConfigStore) RotateOAuth2RefreshToken(ctx context.Context, oldID string, newRT *tables.TableOAuth2RefreshToken) error { + return nil +} +func (m *MockConfigStore) GetOAuth2RefreshTokenByHashAny(ctx context.Context, hash string) (*tables.TableOAuth2RefreshToken, error) { + return nil, configstore.ErrNotFound +} +func (m *MockConfigStore) RevokeOAuth2RefreshTokensByFamilyID(ctx context.Context, familyID string) error { + return nil +} +func (m *MockConfigStore) RevokeOAuth2RefreshTokensByMode(ctx context.Context, bfMode string) error { + return nil +} +func (m *MockConfigStore) SweepOAuth2RefreshTokens(ctx context.Context, revokedOlderThan time.Duration) (int64, error) { + return 0, nil +} +func (m *MockConfigStore) SweepOrphanedOAuth2Clients(ctx context.Context, registeredOlderThan time.Duration) (int64, error) { + return 0, nil +} +func (m *MockConfigStore) ListOAuth2Sessions(ctx context.Context, params configstore.OAuth2SessionsQueryParams) ([]configstore.OAuth2SessionRow, int64, error) { + return nil, 0, nil +} +func (m *MockConfigStore) GetOAuth2SessionByID(ctx context.Context, id string) (*tables.TableOAuth2RefreshToken, error) { + return nil, configstore.ErrNotFound +} +func (m *MockConfigStore) RevokeOAuth2Session(ctx context.Context, id string) error { + return nil +} func (m *MockConfigStore) Ping(ctx context.Context) error { return nil } func (m *MockConfigStore) EncryptPlaintextRows(ctx context.Context) error { return nil } func (m *MockConfigStore) Close(ctx context.Context) error { return nil } @@ -1150,6 +1207,10 @@ func (m *MockConfigStore) UpsertModelParameters(ctx context.Context, params *tab return nil } +func (m *MockConfigStore) UpsertModelParametersBatch(ctx context.Context, params []tables.TableModelParameters, tx ...*gorm.DB) error { + return nil +} + // Provider methods func (m *MockConfigStore) GetProvider(ctx context.Context, provider schemas.ModelProvider) (*tables.TableProvider, error) { return nil, nil @@ -1984,6 +2045,62 @@ func createConfigFile(t *testing.T, dir string, data *ConfigData) { } } +// TestValidateClientConfig_AuthCodeTTL covers the load-time invariant check that +// backs the "fail loudly instead of silently clamp" behavior: an auth_code_ttl +// above the cap is rejected, while nil/zero/in-range/at-cap values pass. +func TestValidateClientConfig_AuthCodeTTL(t *testing.T) { + oauth := func(ttl int) *configstore.ClientConfig { + return &configstore.ClientConfig{OAuth2ServerConfig: &tables.OAuth2ServerConfig{AuthCodeTTL: ttl}} + } + tests := []struct { + name string + cc *configstore.ClientConfig + wantErr bool + }{ + {"no oauth config", &configstore.ClientConfig{}, false}, + {"zero ttl resolves to default at issuance", oauth(0), false}, + {"in-range ttl", oauth(300), false}, + {"exactly at cap", oauth(tables.MaxAuthCodeTTL), false}, + {"one over cap", oauth(tables.MaxAuthCodeTTL + 1), true}, + {"far over cap", oauth(5000), true}, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + err := validateClientConfig(tc.cc) + if tc.wantErr { + require.Error(t, err) + require.Contains(t, err.Error(), "auth_code_ttl") + } else { + require.NoError(t, err) + } + }) + } +} + +// TestLoadConfig_AuthCodeTTLAboveMaxFailsBoot verifies an over-cap auth_code_ttl +// in config.json makes LoadConfig fail rather than silently clamping the value. +func TestLoadConfig_AuthCodeTTLAboveMaxFailsBoot(t *testing.T) { + initTestLogger() + tempDir := createTempDir(t) + createConfigFile(t, tempDir, &ConfigData{ + Client: &configstore.ClientConfig{ + MCPServerAuthMode: tables.MCPServerAuthModeOAuth, + OAuth2ServerConfig: &tables.OAuth2ServerConfig{ + AuthCodeTTL: 5000, + AccessTokenTTL: tables.DefaultAccessTokenTTL, + }, + }, + }) + + ctx := context.Background() + config, err := LoadConfig(ctx, tempDir) + if config != nil { + defer config.Close(ctx) + } + require.Error(t, err) + require.Contains(t, err.Error(), "auth_code_ttl") +} + // TestConfigDataSourceOfTruthDefaultsToSplit verifies omitted source_of_truth uses split mode. func TestConfigDataSourceOfTruthDefaultsToSplit(t *testing.T) { var configData ConfigData @@ -2215,7 +2332,7 @@ func makeVirtualKey(id, name, value string) tables.TableVirtualKey { ID: id, Name: name, Description: "Test virtual key", - Value: value, + Value: *schemas.NewSecretVar(value), IsActive: schemas.Ptr(true), } } @@ -2226,7 +2343,7 @@ func makeVirtualKeyWithTeam(id, name, value, teamID string) tables.TableVirtualK ID: id, Name: name, Description: "Test virtual key with team", - Value: value, + Value: *schemas.NewSecretVar(value), IsActive: schemas.Ptr(true), TeamID: &teamID, } @@ -2238,7 +2355,7 @@ func makeVirtualKeyWithCustomer(id, name, value, customerID string) tables.Table ID: id, Name: name, Description: "Test virtual key with customer", - Value: value, + Value: *schemas.NewSecretVar(value), IsActive: schemas.Ptr(true), CustomerID: &customerID, } @@ -2250,7 +2367,7 @@ func makeVirtualKeyWithProviderConfigs(id, name, value string, providerConfigs [ ID: id, Name: name, Description: "Test virtual key with provider configs", - Value: value, + Value: *schemas.NewSecretVar(value), IsActive: schemas.Ptr(true), ProviderConfigs: providerConfigs, } @@ -3282,7 +3399,53 @@ func TestGenerateKeyHash(t *testing.T) { t.Error("Expected different hash for keys with different BedrockKeyConfig region") } - t.Log("✓ Key hash generation works correctly for all fields including Azure, Vertex, and Bedrock configs") + // BedrockMantleKeyConfig should produce different hash + key9 := schemas.Key{ + ID: "key-1", + Name: "test-key", + Value: *schemas.NewSecretVar("sk-123"), + Models: []string{"gpt-4", "gpt-3.5-turbo"}, + Weight: 1.5, + BedrockMantleKeyConfig: &schemas.BedrockMantleKeyConfig{ + AccessKey: *schemas.NewSecretVar("AKIAIOSFODNN7EXAMPLE"), + SecretKey: *schemas.NewSecretVar("wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY"), + Region: schemas.NewSecretVar(region), + }, + } + + hash9, err := configstore.GenerateKeyHash(key9) + if err != nil { + t.Fatalf("Failed to generate hash: %v", err) + } + + if hash1 == hash9 { + t.Error("Expected different hash for keys with BedrockMantleKeyConfig") + } + + // Different BedrockMantleKeyConfig should produce different hash + key9b := schemas.Key{ + ID: "key-1", + Name: "test-key", + Value: *schemas.NewSecretVar("sk-123"), + Models: []string{"gpt-4", "gpt-3.5-turbo"}, + Weight: 1.5, + BedrockMantleKeyConfig: &schemas.BedrockMantleKeyConfig{ + AccessKey: *schemas.NewSecretVar("AKIAIOSFODNN7EXAMPLE"), + SecretKey: *schemas.NewSecretVar("wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY"), + Region: schemas.NewSecretVar(differentRegion), // Different region + }, + } + + hash9b, err := configstore.GenerateKeyHash(key9b) + if err != nil { + t.Fatalf("Failed to generate hash: %v", err) + } + + if hash9 == hash9b { + t.Error("Expected different hash for keys with different BedrockMantleKeyConfig region") + } + + t.Log("✓ Key hash generation works correctly for all fields including Azure, Vertex, Bedrock, and Bedrock Mantle configs") } // TestProviderHashComparison_MatchingHash tests that DB config is kept when hashes match @@ -7109,7 +7272,7 @@ func TestGenerateVirtualKeyHash(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &teamID, } @@ -7129,7 +7292,7 @@ func TestGenerateVirtualKeyHash(t *testing.T) { ID: "different-id", // Different ID - should be skipped Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &teamID, } @@ -7148,7 +7311,7 @@ func TestGenerateVirtualKeyHash(t *testing.T) { ID: "vk-1", Name: "different-name", // Different name Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &teamID, } @@ -7167,7 +7330,7 @@ func TestGenerateVirtualKeyHash(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_different", // Different value + Value: *schemas.NewSecretVar("vk_different"), // Different value IsActive: schemas.Ptr(true), TeamID: &teamID, } @@ -7186,7 +7349,7 @@ func TestGenerateVirtualKeyHash(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(false), // Different IsActive TeamID: &teamID, } @@ -7200,13 +7363,46 @@ func TestGenerateVirtualKeyHash(t *testing.T) { t.Error("Expected different hash for virtual keys with different IsActive") } + // Setting ExpiresAt should produce a different hash; nil ExpiresAt keeps the original hash + expiry := time.Date(2027, 1, 1, 0, 0, 0, 0, time.UTC) + vkExpiring := tables.TableVirtualKey{ + ID: "vk-1", + Name: "test-vk", + Description: "Test virtual key", + Value: *schemas.NewSecretVar("vk_abc123"), + IsActive: schemas.Ptr(true), + TeamID: &teamID, + ExpiresAt: &expiry, + } + + hashExpiring, err := configstore.GenerateVirtualKeyHash(vkExpiring) + if err != nil { + t.Fatalf("Failed to generate hash: %v", err) + } + + if hash1 == hashExpiring { + t.Error("Expected different hash for virtual keys with ExpiresAt set") + } + + // Different expiry timestamps should produce different hashes + laterExpiry := expiry.Add(time.Hour) + vkExpiring.ExpiresAt = &laterExpiry + hashLaterExpiry, err := configstore.GenerateVirtualKeyHash(vkExpiring) + if err != nil { + t.Fatalf("Failed to generate hash: %v", err) + } + + if hashExpiring == hashLaterExpiry { + t.Error("Expected different hash for virtual keys with different ExpiresAt") + } + // Different TeamID should produce different hash differentTeamID := "team-2" vk6 := tables.TableVirtualKey{ ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &differentTeamID, // Different TeamID } @@ -7225,7 +7421,7 @@ func TestGenerateVirtualKeyHash(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Different description", // Different description - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &teamID, } @@ -7245,7 +7441,7 @@ func TestGenerateVirtualKeyHash(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &teamID, CustomerID: &customerID, // CustomerID set @@ -7266,7 +7462,7 @@ func TestGenerateVirtualKeyHash(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &teamID, CustomerID: &differentCustomerID, // Different CustomerID @@ -7287,7 +7483,7 @@ func TestGenerateVirtualKeyHash(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &teamID, RateLimitID: &rateLimitID, // RateLimitID set @@ -7308,7 +7504,7 @@ func TestGenerateVirtualKeyHash(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &teamID, RateLimitID: &differentRateLimitID, // Different RateLimitID @@ -7335,7 +7531,7 @@ func TestGenerateVirtualKeyHash_WithProviderConfigs(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -7367,7 +7563,7 @@ func TestGenerateVirtualKeyHash_WithProviderConfigs(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -7395,7 +7591,7 @@ func TestGenerateVirtualKeyHash_WithProviderConfigs(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -7432,7 +7628,7 @@ func TestGenerateVirtualKeyHash_AllowAllKeysAndBlacklistedModels(t *testing.T) { return tables.TableVirtualKey{ ID: "vk-1", Name: "test-vk", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -7500,7 +7696,7 @@ func TestGenerateVirtualKeyHash_WithMCPConfigs(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -7526,7 +7722,7 @@ func TestGenerateVirtualKeyHash_WithMCPConfigs(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -7552,7 +7748,7 @@ func TestGenerateVirtualKeyHash_WithMCPConfigs(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -7583,7 +7779,7 @@ func TestVirtualKeyHashComparison_MatchingHash(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &teamID, } @@ -7600,7 +7796,7 @@ func TestVirtualKeyHashComparison_MatchingHash(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &dbTeamID, ConfigHash: fileHash, // Same hash as file @@ -7633,7 +7829,7 @@ func TestVirtualKeyHashComparison_DifferentHash(t *testing.T) { ID: "vk-1", Name: "old-name", // Old name Description: "Old description", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &teamID, } @@ -7650,7 +7846,7 @@ func TestVirtualKeyHashComparison_DifferentHash(t *testing.T) { ID: "vk-1", Name: "new-name", // Updated name Description: "New description", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &fileTeamID, } @@ -7681,7 +7877,7 @@ func TestVirtualKeyHashComparison_VirtualKeyOnlyInDB(t *testing.T) { ID: "vk-dashboard", Name: "dashboard-vk", Description: "Added via dashboard", - Value: "vk_dashboard123", + Value: *schemas.NewSecretVar("vk_dashboard123"), IsActive: schemas.Ptr(true), CustomerID: &customerID, RateLimitID: &rateLimitID, @@ -7699,7 +7895,7 @@ func TestVirtualKeyHashComparison_VirtualKeyOnlyInDB(t *testing.T) { ID: "vk-file", Name: "file-vk", Description: "From config.json", - Value: "vk_file123", + Value: *schemas.NewSecretVar("vk_file123"), IsActive: schemas.Ptr(true), }, } @@ -7729,7 +7925,7 @@ func TestVirtualKeyHashComparison_NewVirtualKey(t *testing.T) { ID: "vk-new", Name: "new-vk", Description: "New virtual key from config.json", - Value: "vk_new123", + Value: *schemas.NewSecretVar("vk_new123"), IsActive: schemas.Ptr(true), TeamID: &teamID, } @@ -7771,7 +7967,7 @@ func TestVirtualKeyHashComparison_OptionalFieldsPresence(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), } @@ -7786,7 +7982,7 @@ func TestVirtualKeyHashComparison_OptionalFieldsPresence(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &teamID, } @@ -7806,7 +8002,7 @@ func TestVirtualKeyHashComparison_OptionalFieldsPresence(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), CustomerID: &customerID, } @@ -7830,7 +8026,7 @@ func TestVirtualKeyHashComparison_OptionalFieldsPresence(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), RateLimitID: &rateLimitID, } @@ -7856,7 +8052,7 @@ func TestVirtualKeyHashComparison_FieldValueChanges(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Base description", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &teamID, } @@ -7919,7 +8115,7 @@ func TestVirtualKeyHashComparison_RoundTrip(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &teamID, RateLimitID: &rateLimitID, @@ -7949,7 +8145,7 @@ func TestVirtualKeyHashComparison_RoundTrip(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), TeamID: &reloadTeamID, RateLimitID: &reloadRateLimitID, @@ -8689,7 +8885,7 @@ func TestSQLite_VirtualKey_HashMismatch_FileSync(t *testing.T) { ID: "vk-1", Name: "modified-name", Description: "Modified description", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), }, } @@ -8742,7 +8938,7 @@ func TestSQLite_VirtualKey_DBOnlyVK_Preserved(t *testing.T) { ID: "vk-dashboard", Name: "dashboard-vk", Description: "Added via dashboard", - Value: "vk_dashboard456", + Value: *schemas.NewSecretVar("vk_dashboard456"), IsActive: schemas.Ptr(true), } dashboardHash, _ := configstore.GenerateVirtualKeyHash(dashboardVK) @@ -8791,7 +8987,7 @@ func TestSQLite_VirtualKey_WithProviderConfigs(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "VK with provider configs", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -8829,7 +9025,7 @@ func TestSQLite_VirtualKey_WithProviderConfigs(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "VK with provider configs", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -8884,7 +9080,7 @@ func TestSQLite_VirtualKey_MergePath_WithProviderConfigs(t *testing.T) { ID: "vk-2", Name: "vk-with-providers", Description: "VK with provider configs added via merge", - Value: "vk_providers456", + Value: *schemas.NewSecretVar("vk_providers456"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -8997,7 +9193,7 @@ func TestSQLite_VirtualKey_MergePath_WithProviderConfigKeys(t *testing.T) { ID: "vk-2", Name: "vk-with-provider-keys", Description: "VK with provider configs referencing keys", - Value: "vk_keys456", + Value: *schemas.NewSecretVar("vk_keys456"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -9055,7 +9251,7 @@ func TestSQLite_VirtualKey_ProviderConfigKeyIDs(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -9072,7 +9268,7 @@ func TestSQLite_VirtualKey_ProviderConfigKeyIDs(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -9104,7 +9300,7 @@ func TestSQLite_VirtualKey_ProviderConfigKeyIDs(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -9145,7 +9341,7 @@ func TestSQLite_VKProviderConfig_NewConfig(t *testing.T) { ID: "vk-1", Name: "vk-with-provider-config", Description: "VK with provider configs", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -9212,7 +9408,7 @@ func TestSQLite_VKProviderConfig_KeyIDsWildcardFlipsAllowAllKeys(t *testing.T) { { ID: "vk-1", Name: "wildcard-vk", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -9246,7 +9442,7 @@ func TestSQLite_VKProviderConfig_KeyIDsWildcardFlipsAllowAllKeys(t *testing.T) { { ID: "vk-1", Name: "wildcard-vk", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -9303,7 +9499,7 @@ func TestSQLite_VKProviderConfig_KeyReference(t *testing.T) { ID: "vk-1", Name: "vk-with-provider-ref", Description: "VK with provider config", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -9355,7 +9551,7 @@ func TestSQLite_VKProviderConfig_HashChangesOnKeyIDChange(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -9373,7 +9569,7 @@ func TestSQLite_VKProviderConfig_HashChangesOnKeyIDChange(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -9405,7 +9601,7 @@ func TestSQLite_VKProviderConfig_HashChangesOnKeyIDChange(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -9438,7 +9634,7 @@ func TestSQLite_VKProviderConfig_WeightAndAllowedModels(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -9454,7 +9650,7 @@ func TestSQLite_VKProviderConfig_WeightAndAllowedModels(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -9470,7 +9666,7 @@ func TestSQLite_VKProviderConfig_WeightAndAllowedModels(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -9513,7 +9709,7 @@ func TestSQLite_VKProviderConfig_WeightAndAllowedModels(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -9601,7 +9797,7 @@ func TestSQLite_FullLifecycle_InitialLoad(t *testing.T) { ID: "vk-1", Name: "test-vk-1", Description: "Test virtual key 1", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), RateLimitID: &rateLimitID, ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ @@ -9616,7 +9812,7 @@ func TestSQLite_FullLifecycle_InitialLoad(t *testing.T) { ID: "vk-2", Name: "test-vk-2", Description: "Test virtual key 2", - Value: "vk_test456", + Value: *schemas.NewSecretVar("vk_test456"), IsActive: schemas.Ptr(true), }, }, @@ -9856,7 +10052,7 @@ func TestSQLite_FullLifecycle_DashboardEdits_ThenFileUnchanged(t *testing.T) { ID: "vk-dashboard", Name: "dashboard-vk", Description: "Added via dashboard", - Value: "vk_dashboard456", + Value: *schemas.NewSecretVar("vk_dashboard456"), IsActive: schemas.Ptr(true), } dashboardHash, _ := configstore.GenerateVirtualKeyHash(dashboardVK) @@ -9919,7 +10115,7 @@ func TestGenerateVirtualKeyHash_MCPConfigChanges(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), } @@ -9928,7 +10124,7 @@ func TestGenerateVirtualKeyHash_MCPConfigChanges(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -9943,7 +10139,7 @@ func TestGenerateVirtualKeyHash_MCPConfigChanges(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -9958,7 +10154,7 @@ func TestGenerateVirtualKeyHash_MCPConfigChanges(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -9973,7 +10169,7 @@ func TestGenerateVirtualKeyHash_MCPConfigChanges(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -10037,7 +10233,7 @@ func TestGenerateVirtualKeyHash_MCPConfigChanges(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -10078,7 +10274,7 @@ func TestSQLite_VirtualKey_WithMCPConfigs(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "VK with MCP config", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), }, } @@ -10167,7 +10363,7 @@ func TestSQLite_VKMCPConfig_Reconciliation(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "VK for MCP reconciliation test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), }, } @@ -10238,7 +10434,7 @@ func TestSQLite_VKMCPConfig_Reconciliation(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "VK for MCP reconciliation test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -10334,7 +10530,7 @@ func TestSQLite_VirtualKey_DashboardProviderConfig_DeletedOnFileChange(t *testin ID: "vk-1", Name: "test-vk", Description: "VK for dashboard provider config preservation test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -10398,7 +10594,7 @@ func TestSQLite_VirtualKey_DashboardProviderConfig_DeletedOnFileChange(t *testin ID: "vk-1", Name: "test-vk", Description: "VK for dashboard provider config preservation test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -10489,7 +10685,7 @@ func TestSQLite_VirtualKey_DashboardMCPConfig_DeletedOnFileChange(t *testing.T) ID: "vk-1", Name: "test-vk", Description: "VK for dashboard MCP config preservation test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), }, } @@ -10574,7 +10770,7 @@ func TestSQLite_VirtualKey_DashboardMCPConfig_DeletedOnFileChange(t *testing.T) ID: "vk-1", Name: "test-vk", Description: "VK for dashboard MCP config preservation test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -10659,7 +10855,7 @@ func TestSQLite_VKMCPConfig_AddRemove(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "VK for add/remove test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), }, } @@ -10695,7 +10891,7 @@ func TestSQLite_VKMCPConfig_AddRemove(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "VK for add/remove test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ {MCPClientID: mcpClient1.ID, ToolsToExecute: []string{"tool1"}}, @@ -10727,7 +10923,7 @@ func TestSQLite_VKMCPConfig_AddRemove(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "VK for add/remove test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ {MCPClientID: mcpClient1.ID, ToolsToExecute: []string{"tool1"}}, @@ -10804,7 +11000,7 @@ func TestSQLite_VKMCPConfig_UpdateTools(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ {MCPClientID: mcpClient.ID, ToolsToExecute: []string{"tool1", "tool2"}}, @@ -10829,7 +11025,7 @@ func TestSQLite_VKMCPConfig_UpdateTools(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ {MCPClientID: mcpClient.ID, ToolsToExecute: []string{"tool3", "tool4", "tool5"}}, // Different tools @@ -10901,7 +11097,7 @@ func TestSQLite_VK_ProviderAndMCPConfigs_Combined(t *testing.T) { ID: "vk-1", Name: "combined-vk", Description: "VK with both provider and MCP configs", - Value: "vk_combined123", + Value: *schemas.NewSecretVar("vk_combined123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -11322,7 +11518,7 @@ func TestGenerateVirtualKeyHash_StableProviderConfigOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -11354,7 +11550,7 @@ func TestGenerateVirtualKeyHash_StableProviderConfigOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -11386,7 +11582,7 @@ func TestGenerateVirtualKeyHash_StableProviderConfigOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -11446,7 +11642,7 @@ func TestGenerateVirtualKeyHash_StableAllowedModelsOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -11464,7 +11660,7 @@ func TestGenerateVirtualKeyHash_StableAllowedModelsOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -11482,7 +11678,7 @@ func TestGenerateVirtualKeyHash_StableAllowedModelsOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -11528,7 +11724,7 @@ func TestGenerateVirtualKeyHash_StableKeyIDsOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -11551,7 +11747,7 @@ func TestGenerateVirtualKeyHash_StableKeyIDsOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -11574,7 +11770,7 @@ func TestGenerateVirtualKeyHash_StableKeyIDsOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -11625,7 +11821,7 @@ func TestGenerateVirtualKeyHash_StableMCPConfigOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -11654,7 +11850,7 @@ func TestGenerateVirtualKeyHash_StableMCPConfigOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -11683,7 +11879,7 @@ func TestGenerateVirtualKeyHash_StableMCPConfigOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -11740,7 +11936,7 @@ func TestGenerateVirtualKeyHash_StableToolsToExecuteOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -11757,7 +11953,7 @@ func TestGenerateVirtualKeyHash_StableToolsToExecuteOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -11774,7 +11970,7 @@ func TestGenerateVirtualKeyHash_StableToolsToExecuteOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), MCPConfigs: []tables.TableVirtualKeyMCPConfig{ { @@ -11819,7 +12015,7 @@ func TestGenerateVirtualKeyHash_StableCombinedOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -11861,7 +12057,7 @@ func TestGenerateVirtualKeyHash_StableCombinedOrdering(t *testing.T) { ID: "vk-1", Name: "test-vk", Description: "Test virtual key", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -12030,7 +12226,7 @@ func TestLoadConfig_Governance_FirstImportUsesMergeIDHandling(t *testing.T) { {Name: "Generated Team"}, }, VirtualKeys: []tables.TableVirtualKey{ - {ID: "vk-explicit", Name: "Explicit VK", Value: "sk-bf-explicit"}, + {ID: "vk-explicit", Name: "Explicit VK", Value: *schemas.NewSecretVar("sk-bf-explicit")}, {Name: "Generated VK"}, }, ComplexityAnalyzerConfig: testFileComplexityAnalyzerConfig(), @@ -12089,11 +12285,11 @@ func TestLoadConfig_Governance_FirstImportUsesMergeIDHandling(t *testing.T) { } require.NotNil(t, explicitVK) require.Equal(t, "vk-explicit", explicitVK.ID) - require.Equal(t, "sk-bf-explicit", explicitVK.Value) + require.Equal(t, "sk-bf-explicit", explicitVK.Value.GetValue()) require.NotNil(t, generatedVK) _, err = uuid.Parse(generatedVK.ID) require.NoError(t, err) - require.True(t, strings.HasPrefix(generatedVK.Value, "sk-bf-")) + require.True(t, strings.HasPrefix(generatedVK.Value.GetValue(), "sk-bf-")) require.NotNil(t, config.GovernanceConfig.ComplexityAnalyzerConfig) require.NotNil(t, govConfig.ComplexityAnalyzerConfig) require.Equal(t, govConfig.ComplexityAnalyzerConfig, config.GovernanceConfig.ComplexityAnalyzerConfig) @@ -14115,7 +14311,7 @@ func TestUpdateGovernanceConfigInStore_RejectsSharedGovernanceIDs(t *testing.T) require.NoError(t, cfg.ConfigStore.CreateVirtualKey(ctx, &tables.TableVirtualKey{ ID: "vk-rl-owner", Name: "vk-rl-owner", - Value: "vk-rl-owner-value", + Value: *schemas.NewSecretVar("vk-rl-owner-value"), IsActive: schemas.Ptr(true), })) require.NoError(t, cfg.ConfigStore.CreateVirtualKeyProviderConfig(ctx, &tables.TableVirtualKeyProviderConfig{ @@ -15245,7 +15441,7 @@ func TestVKProviderConfig_WeightZeroPreserved(t *testing.T) { vk := tables.TableVirtualKey{ ID: "vk-zero-weight", Name: "test-vk", - Value: "vk_test123", + Value: *schemas.NewSecretVar("vk_test123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -15303,7 +15499,7 @@ func TestSQLite_VKProviderConfig_WeightZero_RoundTrip(t *testing.T) { { ID: "vk-zero-weight", Name: "test-vk", - Value: "vk_abc123", + Value: *schemas.NewSecretVar("vk_abc123"), IsActive: schemas.Ptr(true), ProviderConfigs: []tables.TableVirtualKeyProviderConfig{ { @@ -16838,6 +17034,7 @@ var enterpriseSchemaPaths = map[string]bool{ "$schema": true, "access_profiles": true, "audit_logs": true, + "circuit_breaker_config": true, "cluster_config": true, "scim_config": true, "load_balancer_config": true, @@ -16960,6 +17157,7 @@ var excludedSchemaFields = map[string]map[string]bool{ }, "governance": { "business_units": true, // Enterprise feature; not in OSS GovernanceConfig + "roles": true, // Enterprise RBAC role bootstrap; not in OSS GovernanceConfig }, "auth_config": { "disable_auth_on_inference": true, // Deprecated and ignored; kept in schema for backward-compatible config.json validation. Use enforce_auth_on_inference. @@ -17232,6 +17430,7 @@ func TestConfigSchemaSyncTopLevel(t *testing.T) { "$schema": true, "access_profiles": true, "audit_logs": true, + "circuit_breaker_config": true, "cluster_config": true, "scim_config": true, "load_balancer_config": true, @@ -17394,6 +17593,90 @@ func TestResolveFrameworkPricingConfig(t *testing.T) { require.Equal(t, newFileSyncSeconds, *normalizedModelCatalog.PricingSyncInterval) }) + t.Run("model_parameters_url from file overrides db when file changes", func(t *testing.T) { + // DB has a stale model_parameters_url from an earlier startup; config.json + // now points it elsewhere. File wins, mirroring pricing_url. + staleModelParamsURL := "https://stale.example.com/model-parameters.json" + newModelParamsURL := "https://new-file.example.com/model-parameters.json" + storedHash, err := configstore.GenerateFrameworkConfigHash(&fileURL, &staleModelParamsURL, &fileSyncSeconds) + require.NoError(t, err) + dbConfig := &tables.TableFrameworkConfig{ + ID: 11, + PricingURL: &fileURL, + ModelParametersURL: &staleModelParamsURL, + PricingSyncInterval: &fileSyncSeconds, + ConfigHash: storedHash, // hash of OLD file values + } + fileConfig := &framework.FrameworkConfig{ + Pricing: &modelcatalog.Config{ + PricingURL: &fileURL, // unchanged + ModelParametersURL: &newModelParamsURL, // changed + PricingSyncInterval: &fileSyncSeconds, // unchanged + }, + } + + normalizedTable, normalizedModelCatalog, needsDBUpdate := ResolveFrameworkPricingConfig(dbConfig, fileConfig) + require.True(t, needsDBUpdate) + require.Equal(t, newModelParamsURL, *normalizedTable.ModelParametersURL) + require.Equal(t, newModelParamsURL, *normalizedModelCatalog.ModelParametersURL) + }) + + t.Run("model_parameters_url alone triggers hash and overrides db", func(t *testing.T) { + // config.json sets only model_parameters_url (no pricing_url). The file + // hash must still be computed so a changed model_parameters_url is detected. + staleModelParamsURL := "https://stale.example.com/model-parameters.json" + newModelParamsURL := "https://new-file.example.com/model-parameters.json" + storedHash, err := configstore.GenerateFrameworkConfigHash(nil, &staleModelParamsURL, nil) + require.NoError(t, err) + dbConfig := &tables.TableFrameworkConfig{ + ID: 12, + PricingURL: &dbURL, + ModelParametersURL: &staleModelParamsURL, + ConfigHash: storedHash, // hash of OLD file values + } + fileConfig := &framework.FrameworkConfig{ + Pricing: &modelcatalog.Config{ + ModelParametersURL: &newModelParamsURL, + }, + } + + normalizedTable, _, needsDBUpdate := ResolveFrameworkPricingConfig(dbConfig, fileConfig) + require.True(t, needsDBUpdate) + require.Equal(t, newModelParamsURL, *normalizedTable.ModelParametersURL) + }) + + t.Run("unresolved env model_parameters_url does not overwrite db when another field changes", func(t *testing.T) { + // model_parameters_url is an unresolved env.* literal while pricing_url + // changes in the same restart, so fileChanged is true. The stale literal + // must not be persisted over a valid DB value — skip guard preserves it. + rawModelParams := "env.BIFROST_TEST_MODEL_PARAMS_URL_NONEXISTENT_XYZ" + prev, existed := os.LookupEnv("BIFROST_TEST_MODEL_PARAMS_URL_NONEXISTENT_XYZ") + os.Unsetenv("BIFROST_TEST_MODEL_PARAMS_URL_NONEXISTENT_XYZ") + t.Cleanup(func() { + if existed { + os.Setenv("BIFROST_TEST_MODEL_PARAMS_URL_NONEXISTENT_XYZ", prev) + } + }) + validDBModelParams := "https://db.example.com/model-parameters.json" + dbConfig := &tables.TableFrameworkConfig{ + ID: 13, + PricingURL: &dbURL, + ModelParametersURL: &validDBModelParams, + PricingSyncInterval: &dbSyncSeconds, + ConfigHash: "stale-hash", // force fileChanged via the pricing_url diff + } + fileConfig := &framework.FrameworkConfig{ + Pricing: &modelcatalog.Config{ + PricingURL: &fileURL, // changed vs DB + ModelParametersURL: &rawModelParams, + }, + } + + normalizedTable, normalizedModelCatalog, _ := ResolveFrameworkPricingConfig(dbConfig, fileConfig) + require.Equal(t, validDBModelParams, *normalizedTable.ModelParametersURL) + require.Equal(t, validDBModelParams, *normalizedModelCatalog.ModelParametersURL) + }) + t.Run("fallback to file when db fields are missing", func(t *testing.T) { dbConfig := &tables.TableFrameworkConfig{ ID: 3, @@ -18429,11 +18712,11 @@ func TestSQLite_GetVirtualKeysPaginated(t *testing.T) { } vks := []tables.TableVirtualKey{ - {ID: "vk-1", Name: "alpha-key", Value: "val-1", IsActive: schemas.Ptr(true), TeamID: &team1}, - {ID: "vk-2", Name: "beta-key", Value: "val-2", IsActive: schemas.Ptr(true), TeamID: &team2}, - {ID: "vk-3", Name: "alpha-test", Value: "val-3", IsActive: schemas.Ptr(true), CustomerID: &cust1}, - {ID: "vk-4", Name: "gamma-key", Value: "val-4", IsActive: schemas.Ptr(true), CustomerID: &cust2}, - {ID: "vk-5", Name: "delta-key", Value: "val-5", IsActive: schemas.Ptr(true), TeamID: &team1}, + {ID: "vk-1", Name: "alpha-key", Value: *schemas.NewSecretVar("val-1"), IsActive: schemas.Ptr(true), TeamID: &team1}, + {ID: "vk-2", Name: "beta-key", Value: *schemas.NewSecretVar("val-2"), IsActive: schemas.Ptr(true), TeamID: &team2}, + {ID: "vk-3", Name: "alpha-test", Value: *schemas.NewSecretVar("val-3"), IsActive: schemas.Ptr(true), CustomerID: &cust1}, + {ID: "vk-4", Name: "gamma-key", Value: *schemas.NewSecretVar("val-4"), IsActive: schemas.Ptr(true), CustomerID: &cust2}, + {ID: "vk-5", Name: "delta-key", Value: *schemas.NewSecretVar("val-5"), IsActive: schemas.Ptr(true), TeamID: &team1}, } for i := range vks { err := store.CreateVirtualKey(ctx, &vks[i]) diff --git a/transports/bifrost-http/server/oauth2.go b/transports/bifrost-http/server/oauth2.go new file mode 100644 index 00000000000..59176055524 --- /dev/null +++ b/transports/bifrost-http/server/oauth2.go @@ -0,0 +1,91 @@ +package server + +import ( + "context" + "sync" + "time" + + "github.com/maximhq/bifrost/framework/configstore" +) + +// oauth2SweepWorker periodically removes expired authorize requests and old +// revoked refresh tokens from the database, mirroring the pattern used by the +// temp-token sweep worker. +type oauth2SweepWorker struct { + store configstore.ConfigStore + sweepInterval time.Duration + revokedRetention time.Duration + orphanClientGrace time.Duration + stopCh chan struct{} + stopOnce sync.Once + cancel context.CancelFunc +} + +func newOAuth2SweepWorker(store configstore.ConfigStore) *oauth2SweepWorker { + if store == nil { + return nil + } + return &oauth2SweepWorker{ + store: store, + sweepInterval: 10 * time.Minute, + revokedRetention: 30 * 24 * time.Hour, + // Grace before a token-less client is collected. Must exceed the + // authorization code TTL so a client mid-handshake (code issued, not yet + // exchanged) is not swept before it can mint its first token. + orphanClientGrace: time.Hour, + stopCh: make(chan struct{}), + } +} + +func (w *oauth2SweepWorker) start(ctx context.Context) { + runCtx, cancel := context.WithCancel(ctx) + w.cancel = cancel + go w.run(runCtx) +} + +func (w *oauth2SweepWorker) stop() { + w.stopOnce.Do(func() { + // Cancel any in-flight sweep so a blocked DB call unwinds promptly, + // then signal run() to exit its ticker loop. + if w.cancel != nil { + w.cancel() + } + close(w.stopCh) + }) +} + +func (w *oauth2SweepWorker) run(ctx context.Context) { + ticker := time.NewTicker(w.sweepInterval) + defer ticker.Stop() + w.sweep(ctx) + for { + select { + case <-ticker.C: + w.sweep(ctx) + case <-w.stopCh: + return + case <-ctx.Done(): + return + } + } +} + +func (w *oauth2SweepWorker) sweep(ctx context.Context) { + if err := w.store.SweepExpiredOAuth2AuthorizeRequests(ctx); err != nil { + logger.Debug("oauth2 authorize request sweep failed: %v", err) + } + if n, err := w.store.SweepOAuth2RefreshTokens(ctx, w.revokedRetention); err != nil { + logger.Debug("oauth2 refresh token sweep failed: %v", err) + } else if n > 0 { + logger.Debug("oauth2 refresh token sweep removed %d revoked rows", n) + } + // Runs after the refresh token sweep so a client is only collected once its + // tokens have aged out of their retention window, never while they are still + // kept for reuse detection. + if n, err := w.store.SweepOrphanedOAuth2Clients(ctx, w.orphanClientGrace); err != nil { + logger.Debug("oauth2 orphaned client sweep failed: %v", err) + } else if n > 0 { + logger.Debug("oauth2 orphaned client sweep removed %d rows", n) + } +} + diff --git a/transports/bifrost-http/server/server.go b/transports/bifrost-http/server/server.go index b926a50d6b9..f4d7310d9cf 100644 --- a/transports/bifrost-http/server/server.go +++ b/transports/bifrost-http/server/server.go @@ -161,6 +161,17 @@ type BifrostHTTPServer struct { WSTicketStore *handlers.WSTicketStore TempTokens *temptoken.Service TempTokenSweepWorker *temptoken.SweepWorker + OAuth2SweepWorker *oauth2SweepWorker + // OAuth2IdentityResolver scopes a user-mode /mcp request to the user's own + // tools. Optional; wired at server init when user-mode identity resolution + // is available, otherwise left nil (user-mode requests fall back to the + // global server). + OAuth2IdentityResolver handlers.OAuth2IdentityResolver + // ExternalQuotaBudgetResolver supplies budgets/usage for VKs whose + // authoritative usage is tracked outside their own budget rows (enterprise + // access-profile-managed VKs). Optional; wired at server init when available, + // otherwise left nil so the quota endpoint reads the VK's own budget rows. + ExternalQuotaBudgetResolver handlers.ExternalQuotaBudgetResolver wsPool *bfws.Pool } @@ -401,8 +412,8 @@ func (s *BifrostHTTPServer) ReloadVirtualKey(ctx context.Context, id string) (*t } if governanceData := governancePlugin.GetGovernanceStore().GetGovernanceData(ctx); governanceData != nil { for _, existingVK := range governanceData.VirtualKeys { - if existingVK != nil && existingVK.ID == virtualKey.ID && existingVK.Value != "" && existingVK.Value != virtualKey.Value { - s.MCPServerHandler.DeleteVKMCPServer(existingVK.Value) + if existingVK != nil && existingVK.ID == virtualKey.ID && existingVK.Value.IsSet() && existingVK.Value.GetValue() != virtualKey.Value.GetValue() { + s.MCPServerHandler.DeleteVKMCPServer(existingVK.Value.GetValue()) break } } @@ -446,7 +457,7 @@ func (s *BifrostHTTPServer) RemoveVirtualKey(ctx context.Context, id string) err return nil } governancePlugin.GetGovernanceStore().DeleteVirtualKeyInMemory(ctx, id) - s.MCPServerHandler.DeleteVKMCPServer(preloadedVk.Value) + s.MCPServerHandler.DeleteVKMCPServer(preloadedVk.Value.GetValue()) return nil } @@ -1326,7 +1337,17 @@ func (s *BifrostHTTPServer) RegisterInferenceRoutes(ctx context.Context, middlew inferenceHandler := handlers.NewInferenceHandler(s.Client, s.Config) s.IntegrationHandler = handlers.NewIntegrationHandler(s.Client, s.Config, wsResponsesHandler, wsRealtimeHandler, webrtcRealtimeHandler, realtimeClientSecretsHandler) mcpInferenceHandler := handlers.NewMCPInferenceHandler(s.Client, s.Config) - mcpServerHandler, err := handlers.NewMCPServerHandler(ctx, s.Config, s) + // Serve by-ID virtual key lookups on the /mcp JWT auth path from the + // governance in-memory store (avoiding a per-request DB read). Best-effort: + // any store that exposes GetVirtualKeyByID qualifies; otherwise the handler + // falls back to the config store. + var vkCache handlers.VirtualKeyCache + if gp, gerr := s.getGovernancePlugin(); gerr == nil && gp != nil { + if c, ok := gp.GetGovernanceStore().(handlers.VirtualKeyCache); ok { + vkCache = c + } + } + mcpServerHandler, err := handlers.NewMCPServerHandler(ctx, s.Config, s, s.OAuth2IdentityResolver, vkCache) if err != nil { return fmt.Errorf("failed to initialize mcp server handler: %v", err) } @@ -1358,7 +1379,7 @@ func (s *BifrostHTTPServer) RegisterAPIRoutes(ctx context.Context, callbacks Ser } governancePlugin, _ := lib.FindPluginAs[schemas.LLMPlugin](s.Config, governancePluginName) if governancePlugin != nil { - governanceHandler, err = handlers.NewGovernanceHandler(callbacks, s.Config.ConfigStore, govLogManager) + governanceHandler, err = handlers.NewGovernanceHandler(callbacks, s.Config.ConfigStore, govLogManager, s.ExternalQuotaBudgetResolver) if err != nil { return fmt.Errorf("failed to initialize governance handler: %v", err) } @@ -1401,6 +1422,16 @@ func (s *BifrostHTTPServer) RegisterAPIRoutes(ctx context.Context, callbacks Ser promptsHandler := handlers.NewPromptsHandler(s.Config.ConfigStore, promptsReloader) featureFlagsHandler := handlers.NewFeatureFlagsHandler(s.Config.FeatureFlags, s.Config.ConfigStore) // Going ahead with API handlers + oauth2DiscoveryHandler := handlers.NewOAuth2DiscoveryHandler(s.Config) + oauth2IssuanceHandler := handlers.NewOAuth2IssuanceHandler(s.Config, s.TempTokens, s.OAuth2IdentityResolver) + oauth2SessionsHandler := handlers.NewOAuth2SessionsHandler(s.Config) + oauth2ConsentHandler := handlers.NewOAuth2ConsentHandler(s.Config, s.TempTokens, s.OAuth2IdentityResolver) + + oauth2DiscoveryHandler.RegisterRoutes(s.Router, middlewares...) + // No middleware needed for mcp issuance routes, they should be open + oauth2IssuanceHandler.RegisterRoutes(s.Router) + oauth2SessionsHandler.RegisterRoutes(s.Router, middlewares...) + oauth2ConsentHandler.RegisterRoutes(s.Router, middlewares...) healthHandler.RegisterRoutes(s.Router, middlewares...) providerHandler.RegisterRoutes(s.Router, middlewares...) mcpHandler.RegisterRoutes(s.Router, middlewares...) @@ -1741,6 +1772,10 @@ func (s *BifrostHTTPServer) Bootstrap(ctx context.Context) error { if s.TempTokenSweepWorker != nil { s.TempTokenSweepWorker.Start(s.Ctx) } + s.OAuth2SweepWorker = newOAuth2SweepWorker(s.Config.ConfigStore) + if s.OAuth2SweepWorker != nil { + s.OAuth2SweepWorker.start(s.Ctx) + } // Hand the service to the OAuth provider so InitiateUserOAuthFlow mints // a mcp_auth token and embeds it as a URL fragment on the auth-page link. if s.Config.OAuthProvider != nil { @@ -1759,6 +1794,10 @@ func (s *BifrostHTTPServer) Bootstrap(ctx context.Context) error { s.TempTokenSweepWorker.Stop() s.TempTokenSweepWorker = nil } + if s.OAuth2SweepWorker != nil { + s.OAuth2SweepWorker.stop() + s.OAuth2SweepWorker = nil + } return fmt.Errorf("failed to initialize auth middleware: %v", err) } if ctx.Value(schemas.BifrostContextKeyIsEnterprise) == nil { @@ -1770,6 +1809,7 @@ func (s *BifrostHTTPServer) Bootstrap(ctx context.Context) error { if err == nil && semanticCachePlugin != nil { semanticCachePlugin.SetEmbeddingRequestExecutor(s.Client.EmbeddingRequest) } + // Register routes err = s.RegisterAPIRoutes(s.Ctx, s, apiMiddlewares...) if err != nil { @@ -1781,6 +1821,10 @@ func (s *BifrostHTTPServer) Bootstrap(ctx context.Context) error { s.TempTokenSweepWorker.Stop() s.TempTokenSweepWorker = nil } + if s.OAuth2SweepWorker != nil { + s.OAuth2SweepWorker.stop() + s.OAuth2SweepWorker = nil + } return fmt.Errorf("failed to initialize routes: %v", err) } // Registering inference routes @@ -1815,6 +1859,10 @@ func (s *BifrostHTTPServer) Bootstrap(ctx context.Context) error { s.TempTokenSweepWorker.Stop() s.TempTokenSweepWorker = nil } + if s.OAuth2SweepWorker != nil { + s.OAuth2SweepWorker.stop() + s.OAuth2SweepWorker = nil + } return fmt.Errorf("failed to initialize inference routes: %v", err) } // Dial configured MCP clients now that every plugin is registered in the core. @@ -1863,7 +1911,7 @@ func (s *BifrostHTTPServer) Start() error { return fmt.Errorf("failed to create listener on %s: %v", serverAddr, err) } go func() { - logger.Info("successfully started bifrost, serving UI on http://%s:%s", s.Host, s.Port) + logger.Info("successfully started bifrost, serving UI on http://%s", serverAddr) if err := s.Server.Serve(ln); err != nil { errChan <- err } @@ -1914,6 +1962,11 @@ func (s *BifrostHTTPServer) Start() error { logger.Info("stopping temp-token sweep worker...") s.TempTokenSweepWorker.Stop() } + if s.OAuth2SweepWorker != nil { + logger.Info("stopping oauth2 sweep worker...") + s.OAuth2SweepWorker.stop() + s.OAuth2SweepWorker = nil + } if s.devPprofHandler != nil { logger.Info("stopping dev pprof handler...") s.devPprofHandler.Cleanup() diff --git a/transports/changelog.md b/transports/changelog.md index e69de29bb2d..ca21ee2207c 100644 --- a/transports/changelog.md +++ b/transports/changelog.md @@ -0,0 +1,87 @@ +## ✨ Features + +- **DeepSeek Provider** - Added DeepSeek as a first-class provider with dedicated request handling and thinking-mode gating +- **AWS Bedrock Mantle Provider** - Added `bedrock_mantle` as a first-class provider with SigV4 key config, native-Anthropic and OpenAI-compatible routing, DB migration, and UI support +- **OAuth 2.1 Gateway Auth for MCP** - Added a full OAuth2 authorization server for `/mcp`: discovery endpoints, dynamic client registration, authorize/token with PKCE and refresh token rotation, consent page, JWT Bearer authentication, session listing/revocation with sweep worker, OAuth Grants UI, and `mcp_server_auth_mode` config +- **Virtual Key Expiry** - Added an expiry field to virtual keys with governance enforcement +- **ClickHouse Log Store (Beta)** - Added ClickHouse support for the log store, including a hybrid store mode. This feature is in beta and may have some corner cases. +- **IPv6 Support** - Added IPv6 support to the HTTP transport +- **Per-MCP-Server Tool Timeout** - Added per-MCP-server tool execution timeout configuration (thanks [@Purvi09](https://github.com/Purvi09)!) +- **OpenAI Responses Lifecycle APIs** - Added missing OpenAI Responses lifecycle methods with explicit per-verb governance flags (thanks [@17jmumford](https://github.com/17jmumford)!) +- **MCP Clients Filtering & Pagination** - Added connection_type, auth_type, state, virtual_key, and server/client_id filters with pagination and a faceted filter sidebar on the MCP clients page +- **Deprecated Model Marking** - Models are now marked `is_deprecated` in pricing and catalog APIs instead of being filtered out of responses +- **Log Attribution Columns** - Added user, team, customer, and business-unit name columns to the logs list with multi-value attribution cells +- **Latency on Errors** - Error responses now carry latency information +- **Env-Store Virtual Key Values** - Virtual key values now use `schemas.SecretVar`, enabling env-store references +- **Connector Multi-Attribution** - Connectors can now attach multiple teams, customers, and business units +- **Supplemental External Budgets** - Added support for externally resolved supplemental budgets not tracked against a virtual key +- **Cost Recalculation Progress** - Cost recalculation now streams progress via SSE with batch processing +- **Vendor-Prefix Pricing Fallback** - Extended Bedrock vendor-prefix pricing fallback to OpenAI, Google, and xAI models +- **Complexity Router Improvements** - Added stemming alongside exact keyword match and a no-signal fallback to the complexity analyzer +- **MCP VK Header** - Added `x-goog-api-key` as a supported virtual-key header on the MCP auth path + +## 🐞 Fixed + +- **Anthropic Redacted Thinking** - Round-trip `redacted_thinking` blocks on chat completions so tool-use turns with extended thinking replay correctly (thanks [@fus3r](https://github.com/fus3r)!) +- **Bedrock Streaming Block Boundaries** - Emit `contentBlockStop` events on the Bedrock ConverseStream egress (thanks [@fus3r](https://github.com/fus3r)!) +- **Cache Token Accounting** - Report `cached_tokens` as reads only per the OpenAI spec so cache writes are not billed as reads (thanks [@fus3r](https://github.com/fus3r)!) +- **Streaming Retries & Fallbacks** - Clear the per-attempt stream close claim so streaming retries and fallbacks work after SSE-embedded provider errors (thanks [@fus3r](https://github.com/fus3r)!) +- **Governance Team IDs** - Decode URL-encoded team IDs in fetch, update, and delete endpoints (thanks [@nnNyx](https://github.com/nnNyx)!) +- **Semantic Cache Keys** - Resolve semantic cache internal embedding keys like external requests (thanks [@nnNyx](https://github.com/nnNyx)!) +- **Gemini Batch Responses** - Surface Gemini batch inline responses from the response field instead of dest (thanks [@nnNyx](https://github.com/nnNyx)!) +- **Governance Rate-Limit CPU** - Skip O(N) reference refresh on request-time rate-limit and budget reset +- **Tier Cost Calculation** - Evaluate tier costs via input tokens instead of total tokens +- **Cancelled Requests** - Fixed stats and log state for cancelled requests +- **Billing on Failed Streams** - Fixed billing on failed Responses stream requests for Anthropic and Bedrock, and cost for image generation and edit streaming +- **Custom Provider Budgets** - Custom providers with spaces in their names can now set budgets +- **Model Parameters URL** - Honor `model_parameters_url` changes in config.json like `pricing_url` (thanks [@jeremym-tanium](https://github.com/jeremym-tanium)!) +- **Bedrock Truncation Signal** - Signal Bedrock `max_output_tokens` truncation on the Responses API (thanks [@jeremym-tanium](https://github.com/jeremym-tanium)!) +- **MCP Reconnect** - Fixed MCP clients registering as connected with an empty tool set when ListTools fails during startup (thanks [@HackToHell](https://github.com/HackToHell)!) +- **MCP Tool Ordering** - Deterministic MCP tool ordering for prompt cache stability +- **Vertex gs:// Images** - Pass through `gs://` image URLs on Vertex Gemini (thanks [@G-XD](https://github.com/G-XD)!) +- **Hybrid Log Token Usage** - Rebuild token usage from denormalized columns in the hybrid log list (thanks [@G-XD](https://github.com/G-XD)!) +- **Anthropic Files** - Preserve file ID document sources (thanks [@mmacvicar](https://github.com/mmacvicar)!) and forward file IDs and content type on the Anthropic files integration +- **Gemini Upload MIME Type** - Preserve file upload MIME types (thanks [@mmacvicar](https://github.com/mmacvicar)!) +- **Content Logging Bypass** - Sanitize `ErrorDetailsParsed` so raw payloads honor `disable_content_logging` (thanks [@citrocat](https://github.com/citrocat)!), plus error-detail sanitization on the log update path +- **Trace Store Memory Leak** - Sweep orphaned deferred spans in trace store TTL cleanup (thanks [@citrocat](https://github.com/citrocat)!) and complete deferred LLM spans on streaming goroutine exit +- **Claude Code Passthrough Streaming** - Consistent content_block indices for server tools (thanks [@surki](https://github.com/surki)!) +- **Codex Tool Search Round-Trip** - Preserve codex `tool_search_call` and `tool_search_output` input items on the Responses API (thanks [@raghu-nandan-bs](https://github.com/raghu-nandan-bs)!) +- **Gemini Fixes** - Guard tool call config, fix the 2.5-pro thinking budget value, OpenAI-through signature compatibility, and video reference field mapping (thanks [@vojthor](https://github.com/vojthor)!) +- **DeepSeek Thinking** - Convert thinking to disabled when tool choice is required +- **OpenAI Integration** - Propagate `max_tokens` from the OpenAI integration and pass `chunking_strategy` through as an extra param +- **Bedrock Error Types** - Fixed error type setting in all integrations for Bedrock +- **Perplexity Responses** - Fixed Perplexity Responses API compatibility +- **Secret Detection** - Set `SecretTypePlainText` for plain-text JSON and non-prefixed secret values, and check whether virtual key values are secrets +- **Empty Tool Results** - Fixed empty tool call result insertion failures +- **Error Redaction** - Redact decoder details from invalid request payload errors +- **Vertex Idle Timeout** - Fixed idle timeout wiring in the Vertex path +- **Web Fetch** - Assorted web fetch fixes +- **MCP Token Refresh** - Skip background token refresh for disabled or unconfigured MCP clients and exclude terminal-status OAuth configs from the refresh query +- **SSO Login Loop** - Fixed an endless login loop on SSO +- **UI Fixes** - Governance form calendar-aligned toggle gating, dashboard array query params, MCP sessions table scrolling with sticky header, audit logs layout, and model catalog key aliases displayed as model names + +## 🐙 Closed GitHub Issues + +- [#2347](https://github.com/maximhq/bifrost/issues/2347) - MCP tool ordering is non-deterministic, breaking prefix-based prompt caching +- [#3106](https://github.com/maximhq/bifrost/issues/3106) - Governance team delete/fetch fails for SCIM-synced team IDs containing spaces or URL-sensitive characters +- [#3121](https://github.com/maximhq/bifrost/issues/3121) - OpenAI responses.retrieve() not supported +- [#3139](https://github.com/maximhq/bifrost/issues/3139) - Bifrost adds non-standard reasoning/reasoning_details fields to chat completions when using a custom provider for deepseek v4 models +- [#3357](https://github.com/maximhq/bifrost/issues/3357) - Bifrost billing discrepancy for cancelled requests +- [#3951](https://github.com/maximhq/bifrost/issues/3951) - Gemini batch: inline responses (dest.inlinedResponses) are silently dropped, leaving output_file_id null +- [#4262](https://github.com/maximhq/bifrost/issues/4262) - Bedrock ConverseStream egress never emits contentBlockStop (breaks strands streaming) +- [#4314](https://github.com/maximhq/bifrost/issues/4314) - MCP client registered as connected with empty tool set when ListTools fails during connect/reconnect +- [#4402](https://github.com/maximhq/bifrost/issues/4402) - Vertex provider drops image blocks whose URL uses `gs://` scheme +- [#4446](https://github.com/maximhq/bifrost/issues/4446) - Add per MCP server level tool timeout configuration +- [#4679](https://github.com/maximhq/bifrost/issues/4679) - Bedrock Responses API does not signal max_output_tokens truncation +- [#4689](https://github.com/maximhq/bifrost/issues/4689) - Custom providers cannot set budget +- [#4720](https://github.com/maximhq/bifrost/issues/4720) - chunking_strategy is dropped for OpenAI-compatible transcription requests +- [#4721](https://github.com/maximhq/bifrost/issues/4721) - Logs table Tokens column shows N/A when hybrid object storage is enabled +- [#4756](https://github.com/maximhq/bifrost/issues/4756) - semantic_cache internal embedding path bypasses plugin pipeline, causing "no keys found" while direct /v1/embeddings works +- [#4777](https://github.com/maximhq/bifrost/issues/4777) - Image generation stream: completed chunk returns empty output_tokens_details, causing under-billing +- [#4788](https://github.com/maximhq/bifrost/issues/4788) - DeepSeek Anthropic-compatible provider causes "stream closed" error in v1.6.0 (regression from v1.5.16) +- [#4816](https://github.com/maximhq/bifrost/issues/4816) - /v1 chat completions folds cache-write tokens into prompt_tokens_details.cached_tokens +- [#4851](https://github.com/maximhq/bifrost/issues/4851) - v1.6.2 governance rate-limit reset causes high CPU in BumpRateLimitUsage/updateRateLimitReferences +- [#4863](https://github.com/maximhq/bifrost/issues/4863) - model_parameters_url in config.json is ignored after the DB value is set +- [#4868](https://github.com/maximhq/bifrost/issues/4868) - Memory leak: orphaned deferred spans in TraceStore are never TTL-swept +- [#4872](https://github.com/maximhq/bifrost/issues/4872) - Raw request/response payloads bypass disable_content_logging via ErrorDetailsParsed +- [#4942](https://github.com/maximhq/bifrost/issues/4942) - redacted_thinking blocks are dropped on chat completions, breaking tool-use replay with extended thinking diff --git a/transports/config.schema.json b/transports/config.schema.json index 8d455db97d5..50c8f8a766e 100644 --- a/transports/config.schema.json +++ b/transports/config.schema.json @@ -279,6 +279,51 @@ } ], "description": "Public base URL Bifrost uses as the redirect_uri when acting as an OAuth client to upstream MCP servers (Notion, Jira, etc.). Set when Bifrost's callback endpoint is reached via a different URL than its server-side metadata. Supports env var syntax: \"env.MY_VAR\"." + }, + "mcp_server_auth_mode": { + "type": "string", + "enum": ["headers", "both", "oauth"], + "description": "How /mcp authenticates inbound MCP clients. 'headers' (default): VK/api-key/session headers only, discovery disabled. 'both': accepts header credentials and Bifrost-issued JWTs, discovery enabled. 'oauth': Bifrost JWTs only — WARNING: disables VK/header MCP access." + }, + "oauth2_server_config": { + "type": "object", + "properties": { + "issuer_url": { + "anyOf": [ + { "type": "string" }, + { + "type": "object", + "properties": { + "value": { "type": "string" }, + "env_var": { "type": "string" }, + "from_env": { "type": "boolean" } + }, + "additionalProperties": false + } + ], + "description": "Stable public URL advertised as the OAuth2 AS issuer in discovery documents and JWT iss claim. Required for multi-host deployments; single-host deployments can omit this (falls back to request Host header). Supports env var syntax: \"env.MY_VAR\"." + }, + "auth_code_ttl": { + "type": "integer", + "minimum": 1, + "maximum": 900, + "default": 300, + "description": "Lifetime of the single-use authorization code in seconds (default: 300, max: 900 = 15 minutes)." + }, + "access_token_ttl": { + "type": "integer", + "minimum": 1, + "default": 600, + "description": "Lifetime of the issued JWT Bearer token in seconds (default: 600)." + }, + "disable_vk_identity": { + "type": "boolean", + "default": false, + "description": "When true, the OAuth consent flow no longer offers or accepts virtual-key identity and existing virtual-key grants lose access immediately (rejected at request time and unable to refresh). Honored only when an identity provider is configured. Only valid when mcp_server_auth_mode is 'oauth'." + } + }, + "additionalProperties": false, + "description": "OAuth2 authorization server settings for /mcp. Only relevant when mcp_server_auth_mode is 'both' or 'oauth'." } }, "additionalProperties": false @@ -305,6 +350,9 @@ "bedrock": { "$ref": "#/$defs/provider_with_bedrock_config" }, + "bedrock_mantle": { + "$ref": "#/$defs/provider_with_bedrock_mantle_config" + }, "cohere": { "$ref": "#/$defs/provider" }, @@ -353,6 +401,9 @@ "cerebras": { "$ref": "#/$defs/provider" }, + "deepseek": { + "$ref": "#/$defs/provider" + }, "vllm": { "$ref": "#/$defs/provider_with_vllm_config" }, @@ -720,6 +771,11 @@ "description": "Whether the virtual key is active", "default": true }, + "expires_at": { + "type": "string", + "format": "date-time", + "description": "Optional expiry timestamp (RFC3339). Once passed, requests using this virtual key are rejected. Omit for a key that never expires." + }, "calendar_aligned": { "type": "boolean", "description": "Snap all budget resets to calendar boundaries (day, week, month, year)", @@ -1204,7 +1260,7 @@ }, "type": { "type": "string", - "enum": ["sqlite", "postgres"], + "enum": ["sqlite", "postgres", "clickhouse"], "description": "Logs store type" }, "config": { @@ -1332,6 +1388,61 @@ ], "additionalProperties": false } + }, + { + "if": { + "properties": { + "../type": { + "const": "clickhouse" + } + } + }, + "then": { + "properties": { + "host": { + "type": "string", + "description": "ClickHouse host" + }, + "port": { + "type": "string", + "description": "ClickHouse port. Defaults by protocol: native 9000 (9440 TLS), http 8123 (8443 TLS)" + }, + "database": { + "type": "string", + "description": "ClickHouse database name (default: default)" + }, + "username": { + "type": "string", + "description": "ClickHouse username" + }, + "password": { + "type": "string", + "description": "ClickHouse password" + }, + "protocol": { + "type": "string", + "enum": ["native", "http"], + "description": "ClickHouse wire protocol (default: native)" + }, + "secure": { + "type": "boolean", + "description": "Enable TLS (native: secure=true; http: switches to https)", + "default": false + }, + "dial_timeout": { + "type": "integer", + "description": "Connection dial timeout in milliseconds (default: 10000)", + "minimum": 1, + "default": 10000 + }, + "cluster": { + "type": "string", + "description": "Optional cluster name; when set, DDL runs ON CLUSTER with replicated table engines" + } + }, + "required": ["host"], + "additionalProperties": false + } } ] }, @@ -1811,6 +1922,7 @@ "anthropic", "gemini", "bedrock", + "bedrock_mantle", "azure", "cohere", "mistral", @@ -1819,6 +1931,7 @@ "openrouter", "vertex", "cerebras", + "deepseek", "vllm", "parasail", "perplexity", @@ -3342,6 +3455,53 @@ } ] }, + "bedrock_mantle_key": { + "allOf": [ + { + "$ref": "#/$defs/base_key" + }, + { + "type": "object", + "properties": { + "bedrock_mantle_key_config": { + "type": "object", + "properties": { + "access_key": { + "type": "string", + "description": "AWS access key for SigV4 (can use env. prefix)" + }, + "secret_key": { + "type": "string", + "description": "AWS secret key for SigV4 (can use env. prefix)" + }, + "session_token": { + "type": "string", + "description": "AWS session token for temporary credentials (can use env. prefix)" + }, + "region": { + "type": "string", + "description": "AWS region used to build the bedrock-mantle endpoint host" + }, + "role_arn": { + "type": "string", + "description": "AWS IAM role ARN for AssumeRole (can use env. prefix)" + }, + "external_id": { + "type": "string", + "description": "External ID for AssumeRole (can use env. prefix)" + }, + "session_name": { + "type": "string", + "description": "Role session name for AssumeRole (can use env. prefix)" + } + }, + "required": ["region"], + "additionalProperties": false + } + } + } + ] + }, "vllm_key": { "allOf": [ { @@ -3608,6 +3768,45 @@ "required": ["keys"], "additionalProperties": false }, + "provider_with_bedrock_mantle_config": { + "type": "object", + "properties": { + "keys": { + "type": "array", + "items": { + "$ref": "#/$defs/bedrock_mantle_key" + }, + "minItems": 1, + "description": "API keys for this provider" + }, + "network_config": { + "$ref": "#/$defs/network_config" + }, + "concurrency_and_buffer_size": { + "$ref": "#/$defs/concurrency_and_buffer_size" + }, + "proxy_config": { + "$ref": "#/$defs/proxy_config" + }, + "send_back_raw_request": { + "type": "boolean", + "description": "Include raw request in BifrostResponse (default: false)" + }, + "send_back_raw_response": { + "type": "boolean", + "description": "Include raw response in BifrostResponse (default: false)" + }, + "store_raw_request_response": { + "type": "boolean", + "description": "Capture raw request/response for internal logging only; strip from API responses returned to clients (default: false)" + }, + "custom_provider_config": { + "$ref": "#/$defs/custom_provider_config" + } + }, + "required": ["keys"], + "additionalProperties": false + }, "provider_with_vllm_config": { "type": "object", "properties": { @@ -3944,6 +4143,19 @@ } ] }, + "tool_execution_timeout": { + "description": "Per-client override for tool execution timeout. Accepts a Go duration string (e.g. '30s', '2m') or a bare integer treated as seconds. When set, overrides the global tool_manager_config.tool_execution_timeout for this MCP server only. Omit or set to 0 to use the global default.", + "oneOf": [ + { + "type": "string", + "pattern": "^(?:\\d+(?:\\.\\d+)?(?:ns|us|µs|ms|s|m|h))+$" + }, + { + "type": "integer", + "minimum": 0 + } + ] + }, "allowed_extra_headers": { "type": "array", "items": { @@ -5438,6 +5650,7 @@ "azure", "anthropic", "bedrock", + "bedrock_mantle", "cohere", "vertex", "mistral", @@ -5449,6 +5662,7 @@ "parasail", "perplexity", "cerebras", + "deepseek", "gemini", "openrouter", "elevenlabs", @@ -5495,6 +5709,22 @@ "responses_stream": { "type": "boolean" }, + "responses_retrieve": { + "type": "boolean", + "description": "GET stored response by id (OpenAI Responses lifecycle)" + }, + "responses_delete": { + "type": "boolean", + "description": "DELETE stored response (OpenAI Responses lifecycle)" + }, + "responses_cancel": { + "type": "boolean", + "description": "POST cancel in-flight stored response (OpenAI Responses lifecycle)" + }, + "responses_input_items": { + "type": "boolean", + "description": "GET list input items for a stored response (OpenAI Responses lifecycle)" + }, "count_tokens": { "type": "boolean" }, diff --git a/transports/go.mod b/transports/go.mod index aa5b1c4c7a6..0c0115f3a20 100644 --- a/transports/go.mod +++ b/transports/go.mod @@ -10,6 +10,7 @@ require ( github.com/fasthttp/websocket v1.5.12 github.com/go-git/go-billy/v5 v5.9.0 github.com/go-git/go-git/v5 v5.19.1 + github.com/golang-jwt/jwt/v5 v5.3.1 github.com/google/pprof v0.0.0-20251213031049-b05bdaca462f github.com/google/uuid v1.6.0 github.com/klauspost/compress v1.18.6 @@ -123,7 +124,6 @@ require ( github.com/go-openapi/swag/yamlutils v0.25.4 // indirect github.com/go-openapi/validate v0.25.1 // indirect github.com/go-viper/mapstructure/v2 v2.5.0 // indirect - github.com/golang-jwt/jwt/v5 v5.3.1 // indirect github.com/golang/groupcache v0.0.0-20241129210726-2c02b8208cf8 // indirect github.com/google/cel-go v0.28.1 // indirect github.com/google/s2a-go v0.1.9 // indirect diff --git a/transports/version b/transports/version index fdd3be6df54..266146b87cb 100644 --- a/transports/version +++ b/transports/version @@ -1 +1 @@ -1.6.2 +1.6.3 diff --git a/ui/.gitignore b/ui/.gitignore deleted file mode 100644 index e1bfa611088..00000000000 --- a/ui/.gitignore +++ /dev/null @@ -1,43 +0,0 @@ -# See https://help.github.com/articles/ignoring-files/ for more about ignoring files. - -# dependencies -/node_modules -/.pnp -.pnp.* -.yarn/* -!.yarn/patches -!.yarn/plugins -!.yarn/releases -!.yarn/versions - -# testing -/coverage - -# build output -/.next/ -/out/ - -# production -/build - -# misc -.DS_Store -*.pem - -# debug -npm-debug.log* -yarn-debug.log* -yarn-error.log* -.pnpm-debug.log* - -# env files (can opt-in for committing if needed) -.env* - -# vercel -.vercel - -# typescript -*.tsbuildinfo - -# auto-generated TanStack Router route tree -/app/routeTree.gen.ts diff --git a/ui/app/_fallbacks/enterprise/components/circuit-breaker/circuitBreakerView.tsx b/ui/app/_fallbacks/enterprise/components/circuit-breaker/circuitBreakerView.tsx index 144fb4b7eb5..709c7b2cde1 100644 --- a/ui/app/_fallbacks/enterprise/components/circuit-breaker/circuitBreakerView.tsx +++ b/ui/app/_fallbacks/enterprise/components/circuit-breaker/circuitBreakerView.tsx @@ -1,4 +1,4 @@ -import { Zap } from "lucide-react"; +import { CircuitBoard } from "lucide-react"; import ContactUsView from "../views/contactUsView"; export default function CircuitBreakerView() { @@ -6,7 +6,7 @@ export default function CircuitBreakerView() {
} + icon={} title="Unlock circuit breaker for reliable fallbacks" description="This feature is a part of the Bifrost enterprise license. Automatically redirect traffic to a fallback provider when your primary endpoint shows signs of failure." readmeLink="https://docs.getbifrost.ai/enterprise/circuit-breaker" diff --git a/ui/app/_fallbacks/enterprise/components/pii-redactor/piiRedactorProviderView.tsx b/ui/app/_fallbacks/enterprise/components/pii-redactor/piiRedactorProviderView.tsx deleted file mode 100644 index c3506af3589..00000000000 --- a/ui/app/_fallbacks/enterprise/components/pii-redactor/piiRedactorProviderView.tsx +++ /dev/null @@ -1,16 +0,0 @@ -import { ScanEye } from "lucide-react"; -import ContactUsView from "../views/contactUsView"; - -export default function PiiRedactorProviderView() { - return ( -
- } - title="Unlock PII Redaction for better privacy" - description="This feature is a part of the Bifrost enterprise license. We would love to know more about your use case and how we can help you." - readmeLink="https://docs.getbifrost.ai/enterprise/pii-redactor" - /> -
- ); -} \ No newline at end of file diff --git a/ui/app/_fallbacks/enterprise/components/pii-redactor/piiRedactorRulesView.tsx b/ui/app/_fallbacks/enterprise/components/pii-redactor/piiRedactorRulesView.tsx deleted file mode 100644 index 2dd7719f636..00000000000 --- a/ui/app/_fallbacks/enterprise/components/pii-redactor/piiRedactorRulesView.tsx +++ /dev/null @@ -1,16 +0,0 @@ -import { ScanEye } from "lucide-react"; -import ContactUsView from "../views/contactUsView"; - -export default function PiiRedactorRulesView() { - return ( -
- } - title="Unlock PII Redaction for better privacy" - description="This feature is a part of the Bifrost enterprise license. We would love to know more about your use case and how we can help you." - readmeLink="https://docs.getbifrost.ai/enterprise/pii-redactor" - /> -
- ); -} \ No newline at end of file diff --git a/ui/app/_fallbacks/enterprise/lib/contexts/rbacContext.tsx b/ui/app/_fallbacks/enterprise/lib/contexts/rbacContext.tsx index 79064eeb3e1..b54a6b72398 100644 --- a/ui/app/_fallbacks/enterprise/lib/contexts/rbacContext.tsx +++ b/ui/app/_fallbacks/enterprise/lib/contexts/rbacContext.tsx @@ -25,7 +25,6 @@ export enum RbacResource { RBAC = "RBAC", Governance = "Governance", RoutingRules = "RoutingRules", - PIIRedactor = "PIIRedactor", PromptRepository = "PromptRepository", PromptDeploymentStrategy = "PromptDeploymentStrategy", SkillsRepository = "SkillsRepository", @@ -90,4 +89,4 @@ export function useRbacContext() { }; } return context; -} \ No newline at end of file +} diff --git a/ui/app/login/layout.tsx b/ui/app/login/layout.tsx index 8873d3498b3..5e278d2066a 100644 --- a/ui/app/login/layout.tsx +++ b/ui/app/login/layout.tsx @@ -54,6 +54,23 @@ export const Route = createFileRoute("/login")({ // Fetch failed — fall through to login page } if (data && (!data.is_auth_enabled || data.has_valid_token)) { + // If auth is disabled but SSO is configured (restart pending), stay on + // the login page so the user sees the restart notice instead of looping. + if (!data.is_auth_enabled) { + try { + const authTypeRes = await fetch(`${getApiBaseUrl()}/auth/type`, { + credentials: "include", + }); + if (authTypeRes.ok) { + const authType: { type: string } = await authTypeRes.json(); + if (authType.type === "sso") { + return; // SSO configured — show login form with restart notice + } + } + } catch { + // Ignore — fall through to the workspace redirect + } + } throw redirect({ href: postLoginPath }); } }, diff --git a/ui/app/oauth/consent/layout.tsx b/ui/app/oauth/consent/layout.tsx new file mode 100644 index 00000000000..9f665a7e1b4 --- /dev/null +++ b/ui/app/oauth/consent/layout.tsx @@ -0,0 +1,40 @@ +import TempTokenScope from "@/components/tempTokenScope"; +import { ThemeProvider } from "@/components/themeProvider"; +import { ReduxProvider } from "@/lib/store"; +import { NuqsAdapter } from "nuqs/adapters/tanstack-router"; +import { createFileRoute } from "@tanstack/react-router"; +import { Toaster } from "sonner"; +import OAuth2ConsentPage from "./page"; + +// Public OAuth2 consent page — renders outside the dashboard chrome so +// external users who arrive via `claude mcp add` can pick their identity +// without needing a Bifrost dashboard account. +// +// tempTokenScoped: ClientLayout renders MinimalShell and skips the protected +// config fetch when this flag is set. TempTokenScope extracts the `#t=…` +// fragment minted by /oauth2/authorize and attaches it as +// X-Bifrost-Temp-Token on every API call, letting the consent flow APIs +// authenticate the anonymous visitor. +// +// ThemeProvider, ReduxProvider, NuqsAdapter and Toaster are provided here +// directly because this route sits outside /workspace and does not inherit +// them from ClientLayout. +function RouteComponent() { + return ( + + + + + + + + + + + ); +} + +export const Route = createFileRoute("/oauth/consent")({ + staticData: { tempTokenScoped: true }, + component: RouteComponent, +}); diff --git a/ui/app/oauth/consent/page.tsx b/ui/app/oauth/consent/page.tsx new file mode 100644 index 00000000000..7f1324c8f3c --- /dev/null +++ b/ui/app/oauth/consent/page.tsx @@ -0,0 +1,376 @@ +import FullPageLoader from "@/components/fullPageLoader"; +import { Button } from "@/components/ui/button"; +import { Input } from "@/components/ui/input"; +import { Separator } from "@/components/ui/separator"; +import { toast } from "sonner"; +import { + getErrorMessage, + useGetOAuth2ConsentFlowQuery, + useIsAuthEnabledQuery, + useSubmitOAuth2ConsentFlowMutation, +} from "@/lib/store"; +import { + getActiveTempToken, + setActiveTempToken, + setSuppressGlobal401, +} from "@/lib/store/apis/tempToken"; +import { + Fingerprint, + KeyRound, + Loader2, + LogIn, + ShieldCheck, + UserRound, +} from "lucide-react"; +import { useQueryState } from "nuqs"; +import React, { useEffect, useMemo, useState } from "react"; + +export default function OAuth2ConsentPage() { + const [flowId] = useQueryState("flow"); + + if (!flowId) { + return ( + +
+

Missing flow identifier

+

+ This URL is missing the{" "} + flow{" "} + query parameter. Restart the connection from your MCP client. +

+
+
+ ); + } + + return ; +} + +// isSafeRedirect rejects URLs whose scheme could execute script when assigned to +// location.href (javascript:, data:, …) while allowing http(s) and any native +// custom-scheme redirect a client may have registered. +function isSafeRedirect(url: string): boolean { + try { + const proto = new URL(url, window.location.origin).protocol.toLowerCase(); + return !["javascript:", "data:", "vbscript:", "blob:", "file:"].includes(proto); + } catch { + return false; + } +} + +function ConsentView({ flowId }: { flowId: string }) { + const { data: flow, isLoading, isError, error } = useGetOAuth2ConsentFlowQuery(flowId); + const { data: authState } = useIsAuthEnabledQuery(); + const [submitFlow, { isLoading: submitting }] = useSubmitOAuth2ConsentFlowMutation(); + const [vkValue, setVkValue] = useState(""); + const [selectedMode, setSelectedMode] = useState<"vk" | "session" | "user" | null>(null); + + // Restore a temp token persisted across a login round-trip BEFORE sampling + // usingTempToken, so returning to consent after cancelling login (where the + // module-level token was already cleared) still surfaces the sign-in path. + const [usingTempToken] = useState(() => { + if (typeof sessionStorage !== "undefined") { + const stored = sessionStorage.getItem(`oauth2_consent_token_${flowId}`); + if (stored && !getActiveTempToken()) { + setActiveTempToken(stored); + setSuppressGlobal401(true); + sessionStorage.removeItem(`oauth2_consent_token_${flowId}`); + } + } + return getActiveTempToken() !== null; + }); + + const loginHref = useMemo(() => { + const returnPath = `/oauth/consent?flow=${encodeURIComponent(flowId)}`; + return `/login?goto=${encodeURIComponent(returnPath)}`; + }, [flowId]); + + // Persist the active temp token so it can be restored after a login round-trip. + // Kept in an effect (not the memo above) so it runs exactly once per flowId — + // memo computations are not guaranteed to run in React 18 concurrent mode. + useEffect(() => { + if (typeof sessionStorage === "undefined") return; + const currentToken = getActiveTempToken(); + if (currentToken) { + sessionStorage.setItem(`oauth2_consent_token_${flowId}`, currentToken); + } + }, [flowId]); + + const showLoginOption = + usingTempToken && + authState?.is_auth_enabled === true && + authState.has_valid_token === false; + + const handleSubmit = async (mode: "vk" | "session" | "user") => { + setSelectedMode(mode); + try { + const res = await submitFlow({ + flowId, + body: { mode, value: mode === "vk" ? vkValue : undefined }, + }).unwrap(); + // Defence-in-depth: the server validates redirect_uri against the + // registered client, but never hand a javascript:/data: URL to + // location.href — it would execute in this origin. Block dangerous + // schemes while still allowing http(s) and native custom-scheme + // redirects that clients may register. + if (!isSafeRedirect(res.redirect_url)) { + toast.error("Authentication failed", { description: "Invalid redirect URL" }); + setSelectedMode(null); + return; + } + window.location.href = res.redirect_url; + } catch (err) { + toast.error("Authentication failed", { description: getErrorMessage(err) }); + setSelectedMode(null); + } + }; + + if (isLoading) return ; + + if (isError || !flow) { + const status = (error as { status?: number } | undefined)?.status; + if (status === 401) return ; + return ( + +
+

Link unavailable

+

+ This authorization link may have expired or already been used. + Restart the connection from your MCP client to get a fresh link. +

+
+
+ ); + } + + const hasUser = flow.available_modes.includes("user"); + const hasVK = flow.available_modes.includes("vk"); + const hasSession = flow.available_modes.includes("session"); + const hasAnyMode = hasUser || hasVK || hasSession; + const clientName = flow.client_name || "MCP Client"; + + return ( + + {/* Header */} +
+
+ +
+

+ {clientName} wants to connect +

+

+ Choose how you'd like to identify yourself to Bifrost +

+
+ +
+ {/* No mode available — nothing the user can act on here */} + {!hasAnyMode && ( +
+

+ No authentication options available +

+

+ Restart the connection from your MCP client. +

+
+ )} + + {/* User mode — most prominent when logged in */} + {hasUser && flow.logged_in_user && ( +
+
+
+ +
+
+

+ {flow.logged_in_user.name || flow.logged_in_user.id} +

+

Signed-in account

+
+
+ +
+ )} + + {/* User mode — sign in prompt when not logged in */} + {hasUser && !flow.logged_in_user && showLoginOption && ( +
+
+
+ +
+
+

Sign in with your account

+

+ Requires a Bifrost dashboard account +

+
+
+ +
+ )} + + {/* Divider between user and key options */} + {hasUser && (hasVK || hasSession) && ( +
+ + + or + +
+ )} + + {/* VK mode */} + {hasVK && ( +
+
+
+ +
+
+

Virtual Key

+

+ Use a Virtual Key from your Bifrost workspace +

+
+
+ setVkValue(e.target.value)} + onKeyDown={(e) => { + if (e.key === "Enter" && vkValue.trim()) { + void handleSubmit("vk"); + } + }} + disabled={submitting} + className="mb-3" + /> + + {hasUser && ( +

+ If this key is linked to a user account, you'll be asked to sign + in to confirm your identity. +

+ )} +
+ )} + + {hasVK && hasSession && ( +
+ + + or + +
+ )} + + {/* Session mode — de-emphasised, last */} + {hasSession && ( + + )} +
+ + {/* Expiry */} +

+ This link expires {formatExpiry(flow.expires_at)} +

+
+ ); +} + +function formatExpiry(iso: string): string { + const ts = new Date(iso).getTime(); + if (Number.isNaN(ts)) return "soon"; + try { + const diffMs = ts - Date.now(); + if (diffMs < 0) return "soon"; + const mins = Math.floor(diffMs / 60_000); + if (mins < 1) return "in less than a minute"; + if (mins === 1) return "in 1 minute"; + return `in ${mins} minutes`; + } catch { + return "soon"; + } +} + +function Shell({ children }: { children: React.ReactNode }) { + return ( +
+
+ {children} +
+
+ ); +} + +function InvalidLinkView() { + return ( + +
+

+ This link is no longer valid +

+

+ The authorization link has expired, been used already, or had its + token stripped. Restart the connection from your MCP client to get a + fresh link. +

+
+
+ ); +} diff --git a/ui/app/workspace/audit-logs/page.tsx b/ui/app/workspace/audit-logs/page.tsx index 1e651926041..7d64035c497 100644 --- a/ui/app/workspace/audit-logs/page.tsx +++ b/ui/app/workspace/audit-logs/page.tsx @@ -2,7 +2,7 @@ import AuditLogsView from "@enterprise/components/audit-logs/auditLogsView"; export default function AuditLogsPage() { return ( -
+
); diff --git a/ui/app/workspace/complexity-router/page.tsx b/ui/app/workspace/complexity-router/page.tsx index 8538a5db5b8..2b0073e6896 100644 --- a/ui/app/workspace/complexity-router/page.tsx +++ b/ui/app/workspace/complexity-router/page.tsx @@ -37,7 +37,7 @@ import { z } from "zod"; type TierBoundaryKey = keyof TierBoundaries; -const COMPLEXITY_ROUTER_DOCS_URL = "https://docs.getbifrost.ai/features/governance/complexity-router"; +const KEYWORD_COLLAPSED_LIMIT = 8; // Four progressive shades of --primary: faintest → full const P1 = "color-mix(in oklch, var(--primary) 30%, transparent)"; @@ -199,7 +199,7 @@ function TierSpectrumBar({ boundaries }: { boundaries: TierBoundaries }) { return (
-
+
{segments.map(({ tier, width, color }) => (
-
+ {/* ── Page header ── */}

Complexity Router

Tune how incoming requests are classified into four tiers. Thresholds and keyword lists feed the{" "} - complexity_tier field that routing rules can target. + complexity_tier field that routing rules can + target.

{/* ── Complexity Spectrum ── */} -
+

Complexity Spectrum

@@ -404,7 +405,7 @@ export default function ComplexityRouterPage() { }); return ( -
+
{/* Tier transition label */}
@@ -470,7 +471,7 @@ export default function ComplexityRouterPage() { const fieldError = keywordErrors?.[key as KeywordListKey]; const errorId = `keywords-${key}-error`; return ( -
+
{submitError}
)} {/* ── Action footer ── */} -
+
+
+ + + Cancelled + + {(data.cancelled ?? 0).toLocaleString()} +
); @@ -64,6 +72,7 @@ function LogVolumeChartImpl({ data, chartType, startTime, endTime }: LogVolumeCh return data.buckets.map((bucket, index) => ({ ...bucket, + cancelled: bucket.cancelled ?? 0, index, formattedTime: formatTimestamp(bucket.timestamp, data.bucket_size_seconds), })); @@ -119,6 +128,15 @@ function LogVolumeChartImpl({ data, chartType, startTime, endTime }: LogVolumeCh stackId="requests" fill={CHART_COLORS.error} fillOpacity={0.9} + radius={[0, 0, 0, 0]} + barSize={30} + /> + @@ -163,6 +181,15 @@ function LogVolumeChartImpl({ data, chartType, startTime, endTime }: LogVolumeCh fill={CHART_COLORS.error} fillOpacity={0.7} /> + )} @@ -170,4 +197,4 @@ function LogVolumeChartImpl({ data, chartType, startTime, endTime }: LogVolumeCh ); } -export const LogVolumeChart = memo(LogVolumeChartImpl); \ No newline at end of file +export const LogVolumeChart = memo(LogVolumeChartImpl); diff --git a/ui/app/workspace/dashboard/components/charts/modelUsageChart.tsx b/ui/app/workspace/dashboard/components/charts/modelUsageChart.tsx index f23c3f7dd27..9c6cd8d4763 100644 --- a/ui/app/workspace/dashboard/components/charts/modelUsageChart.tsx +++ b/ui/app/workspace/dashboard/components/charts/modelUsageChart.tsx @@ -77,6 +77,15 @@ function CustomTooltip({ active, payload, selectedModel, displayModels }: any) { {(data.by_model?.[selectedModel]?.error || 0).toLocaleString()}
+
+ + + Cancelled + + + {(data.by_model?.[selectedModel]?.cancelled || 0).toLocaleString()} + +
)}
@@ -124,10 +133,11 @@ function ModelUsageChartImpl({ data, chartType, startTime, endTime, selectedMode } }); } else { - // For specific model, show success/error breakdown + // For specific model, show success/error/cancelled breakdown const stats = bucket.by_model?.[selectedModel]; item.success = stats?.success || 0; item.error = stats?.error || 0; + item.cancelled = stats?.cancelled || 0; } return item; }); @@ -203,6 +213,15 @@ function ModelUsageChartImpl({ data, chartType, startTime, endTime, selectedMode stackId="status" fill={CHART_COLORS.error} fillOpacity={0.9} + radius={[0, 0, 0, 0]} + barSize={30} + /> + @@ -271,6 +290,15 @@ function ModelUsageChartImpl({ data, chartType, startTime, endTime, selectedMode fill={CHART_COLORS.error} fillOpacity={0.7} /> + )} @@ -280,4 +308,4 @@ function ModelUsageChartImpl({ data, chartType, startTime, endTime, selectedMode ); } -export const ModelUsageChart = memo(ModelUsageChartImpl); \ No newline at end of file +export const ModelUsageChart = memo(ModelUsageChartImpl); diff --git a/ui/app/workspace/dashboard/components/overviewTab.tsx b/ui/app/workspace/dashboard/components/overviewTab.tsx index ebcb7d8065e..fb9a3d72aa7 100644 --- a/ui/app/workspace/dashboard/components/overviewTab.tsx +++ b/ui/app/workspace/dashboard/components/overviewTab.tsx @@ -170,6 +170,10 @@ function OverviewTabImpl({ Error + + + Cancelled +
} controls={ @@ -360,6 +364,10 @@ function OverviewTabImpl({ Error + + + Cancelled + )}
@@ -421,4 +429,4 @@ function OverviewTabImpl({ ); } -export const OverviewTab = memo(OverviewTabImpl); \ No newline at end of file +export const OverviewTab = memo(OverviewTabImpl); diff --git a/ui/app/workspace/dashboard/page.tsx b/ui/app/workspace/dashboard/page.tsx index 2418129d9c7..122bbe06f98 100644 --- a/ui/app/workspace/dashboard/page.tsx +++ b/ui/app/workspace/dashboard/page.tsx @@ -3,12 +3,13 @@ import { DateTimePickerWithRange } from "@/components/ui/datePickerWithRange"; import { ScrollArea } from "@/components/ui/scrollArea"; import { Tabs, TabsContent, TabsList, TabsTrigger } from "@/components/ui/tabs"; import { useTimezonePreference } from "@/lib/hooks/useTimezonePreference"; +import { parseAsSafeArrayOf } from "@/lib/queryParamsParser"; import { useGetMCPAvailableFilterDataQuery } from "@/lib/store"; import type { LogFilters, MCPToolLogFilters } from "@/lib/types/logs"; import { dateUtils } from "@/lib/types/logs"; import { getRangeForPeriod, TIME_PERIODS } from "@/lib/utils/timeRange"; import { useLocation } from "@tanstack/react-router"; -import { parseAsInteger, parseAsString, useQueryStates } from "nuqs"; +import { parseAsBoolean, parseAsInteger, parseAsString, useQueryStates } from "nuqs"; import { useCallback, useMemo, useRef, useState } from "react"; import { type ChartType } from "./components/charts/chartTypeToggle"; import { ModelFilterSelect } from "./components/charts/modelFilterSelect"; @@ -22,8 +23,6 @@ import type { DashboardData } from "./utils/exportUtils"; const toChartType = (value: string): ChartType => (value === "line" ? "line" : "bar"); -const parseCsvParam = (value: string): string[] => (value ? value.split(",").filter(Boolean) : []); - export default function DashboardPage() { // MCP filter data const { data: mcpFilterData } = useGetMCPAvailableFilterDataQuery(); @@ -41,16 +40,16 @@ export default function DashboardPage() { start_time: parseAsInteger.withDefault(defaultTimeRange.startTime), end_time: parseAsInteger.withDefault(defaultTimeRange.endTime), tab: parseAsString.withDefault("overview"), - virtual_key_ids: parseAsString.withDefault(""), - providers: parseAsString.withDefault(""), - models: parseAsString.withDefault(""), - selected_key_ids: parseAsString.withDefault(""), - objects: parseAsString.withDefault(""), - status: parseAsString.withDefault(""), - routing_rule_ids: parseAsString.withDefault(""), - routing_engine_used: parseAsString.withDefault(""), - stop_reasons: parseAsString.withDefault(""), - missing_cost_only: parseAsString.withDefault("false"), + virtual_key_ids: parseAsSafeArrayOf.withDefault([]), + providers: parseAsSafeArrayOf.withDefault([]), + models: parseAsSafeArrayOf.withDefault([]), + selected_key_ids: parseAsSafeArrayOf.withDefault([]), + objects: parseAsSafeArrayOf.withDefault([]), + status: parseAsSafeArrayOf.withDefault([]), + routing_rule_ids: parseAsSafeArrayOf.withDefault([]), + routing_engine_used: parseAsSafeArrayOf.withDefault([]), + stop_reasons: parseAsSafeArrayOf.withDefault([]), + missing_cost_only: parseAsBoolean.withDefault(false), metadata_filters: parseAsString.withDefault(""), volume_chart: parseAsString.withDefault("bar"), token_chart: parseAsString.withDefault("bar"), @@ -70,11 +69,11 @@ export default function DashboardPage() { mcp_tool_names: parseAsString.withDefault(""), mcp_server_labels: parseAsString.withDefault(""), parent_request_id: parseAsString.withDefault(""), - user_ids: parseAsString.withDefault(""), - team_ids: parseAsString.withDefault(""), - customer_ids: parseAsString.withDefault(""), - business_unit_ids: parseAsString.withDefault(""), - aliases: parseAsString.withDefault(""), + user_ids: parseAsSafeArrayOf.withDefault([]), + team_ids: parseAsSafeArrayOf.withDefault([]), + customer_ids: parseAsSafeArrayOf.withDefault([]), + business_unit_ids: parseAsSafeArrayOf.withDefault([]), + aliases: parseAsSafeArrayOf.withDefault([]), }, { history: "push", @@ -82,17 +81,13 @@ export default function DashboardPage() { }, ); - // Parse filter arrays from URL state - const selectedProviders = useMemo(() => parseCsvParam(urlState.providers), [urlState.providers]); - const selectedModels = useMemo(() => parseCsvParam(urlState.models), [urlState.models]); - const selectedKeyIds = useMemo(() => parseCsvParam(urlState.selected_key_ids), [urlState.selected_key_ids]); - const selectedVirtualKeyIds = useMemo(() => parseCsvParam(urlState.virtual_key_ids), [urlState.virtual_key_ids]); - const selectedTypes = useMemo(() => parseCsvParam(urlState.objects), [urlState.objects]); - const selectedStatuses = useMemo(() => parseCsvParam(urlState.status), [urlState.status]); - const selectedRoutingRuleIds = useMemo(() => parseCsvParam(urlState.routing_rule_ids), [urlState.routing_rule_ids]); - const selectedRoutingEngines = useMemo(() => parseCsvParam(urlState.routing_engine_used), [urlState.routing_engine_used]); - const selectedStopReasons = useMemo(() => parseCsvParam(urlState.stop_reasons), [urlState.stop_reasons]); - const missingCostOnly = useMemo(() => urlState.missing_cost_only === "true", [urlState.missing_cost_only]); + // Parse string-backed MCP filter values from URL state + const selectedMcpToolNames = useMemo(() => (urlState.mcp_tool_names ? [urlState.mcp_tool_names] : []), [urlState.mcp_tool_names]); + const selectedMcpServerLabels = useMemo( + () => (urlState.mcp_server_labels ? [urlState.mcp_server_labels] : []), + [urlState.mcp_server_labels], + ); + const metadataFilters = useMemo(() => { if (!urlState.metadata_filters) return undefined; try { @@ -102,15 +97,6 @@ export default function DashboardPage() { } }, [urlState.metadata_filters]); - const selectedMcpToolNames = useMemo(() => parseCsvParam(urlState.mcp_tool_names), [urlState.mcp_tool_names]); - const selectedMcpServerLabels = useMemo(() => parseCsvParam(urlState.mcp_server_labels), [urlState.mcp_server_labels]); - - const selectedUserIds = useMemo(() => parseCsvParam(urlState.user_ids), [urlState.user_ids]); - const selectedTeamIds = useMemo(() => parseCsvParam(urlState.team_ids), [urlState.team_ids]); - const selectedCustomerIds = useMemo(() => parseCsvParam(urlState.customer_ids), [urlState.customer_ids]); - const selectedBusinessUnitIds = useMemo(() => parseCsvParam(urlState.business_unit_ids), [urlState.business_unit_ids]); - const selectedAliases = useMemo(() => parseCsvParam(urlState.aliases), [urlState.aliases]); - const filters: LogFilters = useMemo( () => ({ ...(urlState.period @@ -119,54 +105,54 @@ export default function DashboardPage() { start_time: dateUtils.toISOString(urlState.start_time), end_time: dateUtils.toISOString(urlState.end_time), }), - ...(selectedProviders.length > 0 && { providers: selectedProviders }), - ...(selectedModels.length > 0 && { models: selectedModels }), - ...(selectedKeyIds.length > 0 && { selected_key_ids: selectedKeyIds }), - ...(selectedVirtualKeyIds.length > 0 && { - virtual_key_ids: selectedVirtualKeyIds, + ...(urlState.providers.length > 0 && { providers: urlState.providers }), + ...(urlState.models.length > 0 && { models: urlState.models }), + ...(urlState.selected_key_ids.length > 0 && { selected_key_ids: urlState.selected_key_ids }), + ...(urlState.virtual_key_ids.length > 0 && { + virtual_key_ids: urlState.virtual_key_ids, }), - ...(selectedTypes.length > 0 && { objects: selectedTypes }), - ...(selectedStatuses.length > 0 && { status: selectedStatuses }), - ...(selectedRoutingRuleIds.length > 0 && { - routing_rule_ids: selectedRoutingRuleIds, + ...(urlState.objects.length > 0 && { objects: urlState.objects }), + ...(urlState.status.length > 0 && { status: urlState.status }), + ...(urlState.routing_rule_ids.length > 0 && { + routing_rule_ids: urlState.routing_rule_ids, }), - ...(selectedRoutingEngines.length > 0 && { - routing_engine_used: selectedRoutingEngines, + ...(urlState.routing_engine_used.length > 0 && { + routing_engine_used: urlState.routing_engine_used, }), - ...(selectedStopReasons.length > 0 && { stop_reasons: selectedStopReasons }), - ...(missingCostOnly && { missing_cost_only: true }), + ...(urlState.stop_reasons.length > 0 && { stop_reasons: urlState.stop_reasons }), + ...(urlState.missing_cost_only && { missing_cost_only: true }), ...(metadataFilters && Object.keys(metadataFilters).length > 0 && { metadata_filters: metadataFilters, }), ...(urlState.parent_request_id && { parent_request_id: urlState.parent_request_id }), - ...(selectedUserIds.length > 0 && { user_ids: selectedUserIds }), - ...(selectedTeamIds.length > 0 && { team_ids: selectedTeamIds }), - ...(selectedCustomerIds.length > 0 && { customer_ids: selectedCustomerIds }), - ...(selectedBusinessUnitIds.length > 0 && { business_unit_ids: selectedBusinessUnitIds }), - ...(selectedAliases.length > 0 && { aliases: selectedAliases }), + ...(urlState.user_ids.length > 0 && { user_ids: urlState.user_ids }), + ...(urlState.team_ids.length > 0 && { team_ids: urlState.team_ids }), + ...(urlState.customer_ids.length > 0 && { customer_ids: urlState.customer_ids }), + ...(urlState.business_unit_ids.length > 0 && { business_unit_ids: urlState.business_unit_ids }), + ...(urlState.aliases.length > 0 && { aliases: urlState.aliases }), }), [ urlState.period, urlState.start_time, urlState.end_time, urlState.parent_request_id, - selectedProviders, - selectedModels, - selectedKeyIds, - selectedVirtualKeyIds, - selectedTypes, - selectedStatuses, - selectedRoutingRuleIds, - selectedRoutingEngines, - selectedStopReasons, - missingCostOnly, + urlState.providers, + urlState.models, + urlState.selected_key_ids, + urlState.virtual_key_ids, + urlState.objects, + urlState.status, + urlState.routing_rule_ids, + urlState.routing_engine_used, + urlState.stop_reasons, + urlState.missing_cost_only, metadataFilters, - selectedUserIds, - selectedTeamIds, - selectedCustomerIds, - selectedBusinessUnitIds, - selectedAliases, + urlState.user_ids, + urlState.team_ids, + urlState.customer_ids, + urlState.business_unit_ids, + urlState.aliases, ], ); @@ -184,9 +170,9 @@ export default function DashboardPage() { ...(selectedMcpServerLabels.length > 0 && { server_labels: selectedMcpServerLabels, }), - ...(selectedStatuses.length > 0 && { status: selectedStatuses }), - ...(selectedVirtualKeyIds.length > 0 && { - virtual_key_ids: selectedVirtualKeyIds, + ...(urlState.status.length > 0 && { status: urlState.status }), + ...(urlState.virtual_key_ids.length > 0 && { + virtual_key_ids: urlState.virtual_key_ids, }), }), [ @@ -195,8 +181,8 @@ export default function DashboardPage() { urlState.end_time, selectedMcpToolNames, selectedMcpServerLabels, - selectedStatuses, - selectedVirtualKeyIds, + urlState.status, + urlState.virtual_key_ids, ], ); @@ -290,7 +276,7 @@ export default function DashboardPage() { [setUrlState], ); - // Adapter: converts a full LogFilters object to dashboard's CSV-based URL state + // Adapter: converts a full LogFilters object to dashboard URL state const setFilters = useCallback( (newFilters: LogFilters) => { const newStartTime = newFilters.start_time ? dateUtils.toUnixTimestamp(new Date(newFilters.start_time)) : undefined; @@ -300,30 +286,29 @@ export default function DashboardPage() { ...(timeChanged && { period: "" }), start_time: newStartTime, end_time: newEndTime, - period: urlState.period, - providers: (newFilters.providers || []).join(","), - models: (newFilters.models || []).join(","), - selected_key_ids: (newFilters.selected_key_ids || []).join(","), - virtual_key_ids: (newFilters.virtual_key_ids || []).join(","), - objects: (newFilters.objects || []).join(","), - status: (newFilters.status || []).join(","), - routing_rule_ids: (newFilters.routing_rule_ids || []).join(","), - routing_engine_used: (newFilters.routing_engine_used || []).join(","), - stop_reasons: (newFilters.stop_reasons || []).join(","), - missing_cost_only: String(newFilters.missing_cost_only ?? false), + providers: newFilters.providers || [], + models: newFilters.models || [], + selected_key_ids: newFilters.selected_key_ids || [], + virtual_key_ids: newFilters.virtual_key_ids || [], + objects: newFilters.objects || [], + status: newFilters.status || [], + routing_rule_ids: newFilters.routing_rule_ids || [], + routing_engine_used: newFilters.routing_engine_used || [], + stop_reasons: newFilters.stop_reasons || [], + missing_cost_only: newFilters.missing_cost_only ?? false, metadata_filters: newFilters.metadata_filters && Object.keys(newFilters.metadata_filters).length > 0 ? JSON.stringify(newFilters.metadata_filters) : "", parent_request_id: newFilters.parent_request_id || "", - user_ids: (newFilters.user_ids || []).join(","), - team_ids: (newFilters.team_ids || []).join(","), - customer_ids: (newFilters.customer_ids || []).join(","), - business_unit_ids: (newFilters.business_unit_ids || []).join(","), - aliases: (newFilters.aliases || []).join(","), + user_ids: newFilters.user_ids || [], + team_ids: newFilters.team_ids || [], + customer_ids: newFilters.customer_ids || [], + business_unit_ids: newFilters.business_unit_ids || [], + aliases: newFilters.aliases || [], }); }, - [setUrlState, urlState.start_time, urlState.end_time, urlState.period], + [setUrlState, urlState.start_time, urlState.end_time], ); // Date range for picker @@ -690,4 +675,4 @@ export default function DashboardPage() {
); -} \ No newline at end of file +} diff --git a/ui/app/workspace/dashboard/utils/chartUtils.ts b/ui/app/workspace/dashboard/utils/chartUtils.ts index 12019a3ce7f..11f431720ce 100644 --- a/ui/app/workspace/dashboard/utils/chartUtils.ts +++ b/ui/app/workspace/dashboard/utils/chartUtils.ts @@ -108,9 +108,10 @@ export const CHART_HEADER_CONTROLS_CLASS = "flex items-center justify-end gap-2" export const CHART_COLORS = { success: "#10b981", // emerald-500 error: "#ef4444", // red-500 + cancelled: "#a1a1aa", // zinc-400 promptTokens: "#3b82f6", // blue-500 completionTokens: "#10b981", // emerald-500 totalTokens: "#8b5cf6", // violet-500 cost: "#f59e0b", // amber-500 cachedReadTokens: "#06b6d4", // cyan-500 -}; \ No newline at end of file +}; diff --git a/ui/app/workspace/dashboard/utils/exportUtils.ts b/ui/app/workspace/dashboard/utils/exportUtils.ts index dc742138ed2..f24aaa74e30 100644 --- a/ui/app/workspace/dashboard/utils/exportUtils.ts +++ b/ui/app/workspace/dashboard/utils/exportUtils.ts @@ -24,8 +24,8 @@ import type { type CSVData = { headers: string[]; rows: unknown[][] }; export function overviewVolumeToCSV(data: LogsHistogramResponse | null): CSVData { - const headers = ["Timestamp", "Total Requests", "Success", "Error"]; - const rows = (data?.buckets ?? []).map((b) => [b.timestamp, b.count, b.success, b.error]); + const headers = ["Timestamp", "Total Requests", "Success", "Error", "Cancelled"]; + const rows = (data?.buckets ?? []).map((b) => [b.timestamp, b.count, b.success, b.error, b.cancelled ?? 0]); return { headers, rows }; } @@ -44,13 +44,13 @@ export function overviewCostToCSV(data: CostHistogramResponse | null): CSVData { export function overviewModelUsageToCSV(data: ModelHistogramResponse | null): CSVData { const models = data?.models ?? []; - const modelHeaders = models.flatMap((m) => [`${m} Total`, `${m} Success`, `${m} Error`]); + const modelHeaders = models.flatMap((m) => [`${m} Total`, `${m} Success`, `${m} Error`, `${m} Cancelled`]); const headers = ["Timestamp", ...modelHeaders]; const rows = (data?.buckets ?? []).map((b) => [ b.timestamp, ...models.flatMap((m) => { const stats = b.by_model?.[m]; - return [stats?.total ?? 0, stats?.success ?? 0, stats?.error ?? 0]; + return [stats?.total ?? 0, stats?.success ?? 0, stats?.error ?? 0, stats?.cancelled ?? 0]; }), ]); return { headers, rows }; @@ -269,4 +269,4 @@ export function getCSVSections(data: DashboardData, tab: ExportTab): { name: str } return sections; -} \ No newline at end of file +} diff --git a/ui/app/workspace/logs/views/columns.tsx b/ui/app/workspace/logs/views/columns.tsx index 05a48f72da0..a42f4537e38 100644 --- a/ui/app/workspace/logs/views/columns.tsx +++ b/ui/app/workspace/logs/views/columns.tsx @@ -1,5 +1,4 @@ import { formatCost, formatLatency } from "@/app/workspace/dashboard/utils/chartUtils"; -import { formatCompactNumber } from "@/lib/utils/numbers"; import { Badge } from "@/components/ui/badge"; import { Button } from "@/components/ui/button"; import { DropdownMenu, DropdownMenuContent, DropdownMenuItem, DropdownMenuTrigger } from "@/components/ui/dropdownMenu"; @@ -7,6 +6,7 @@ import { ProviderIconType, RenderProviderIcon } from "@/lib/constants/icons"; import { getProviderLabel, ProviderName, RequestTypeColors, RequestTypeLabels, Status, StatusBarColors } from "@/lib/constants/logs"; import { ChatMessageContent, LogEntry, ResponsesMessageContentBlock } from "@/lib/types/logs"; import { cn } from "@/lib/utils"; +import { formatCompactNumber } from "@/lib/utils/numbers"; import { ColumnDef } from "@tanstack/react-table"; import { format, formatDistanceToNow } from "date-fns"; import { ArrowUpDown, MoreHorizontal, Trash2 } from "lucide-react"; @@ -178,7 +178,7 @@ export function LogMessageCell({ log, contentClassName = "max-w-full" }: { log: )} {realtimeMessages && - (realtimeMessages.tool || realtimeMessages.user || realtimeMessages.assistantToolCall || realtimeMessages.assistant) ? ( + (realtimeMessages.tool || realtimeMessages.user || realtimeMessages.assistantToolCall || realtimeMessages.assistant) ? (
{realtimeMessages.tool ?
Tool Result: {realtimeMessages.tool}
: null} {realtimeMessages.user ?
User: {realtimeMessages.user}
: null} @@ -199,6 +199,53 @@ export function LogMessageCell({ log, contentClassName = "max-w-full" }: { log: ); } +const MAX_ATTRIBUTION_LINES = 1; + +// AttributionCell resolves an attribution value using a plural-first fallback: +// plural names -> singular name -> plural ids -> singular id. When a plural +// (array) source is used, values render one per line, capped at +// MAX_ATTRIBUTION_LINES with a "+N more" indicator for the remainder. +function AttributionCell({ + names, + name, + ids, + id, +}: { + names?: string[]; + name?: string | null; + ids?: string[]; + id?: string | null; +}) { + let values: string[] = []; + if (Array.isArray(names) && names.filter(Boolean).length > 0) { + values = names.filter(Boolean); + } else if (name) { + values = [name]; + } else if (Array.isArray(ids) && ids.filter(Boolean).length > 0) { + values = ids.filter(Boolean); + } else if (id) { + values = [id]; + } + + if (values.length === 0) { + return
-
; + } + + const visible = values.slice(0, MAX_ATTRIBUTION_LINES); + const remaining = values.length - visible.length; + + return ( +
+ {visible.map((value, index) => ( + + {value} + + ))} + {remaining > 0 && +{remaining} more} +
+ ); +} + export const createColumns = ( onDelete: (log: LogEntry) => void, hasDeleteAccess = true, @@ -367,44 +414,63 @@ export const createColumns = ( }, ]; - const attributionCell = (value?: string | null) =>
{value || "-"}
; - const attributionColumns: ColumnDef[] = [ { id: "virtual_key", header: "Virtual Key", size: 170, - cell: ({ row }) => attributionCell(row.original.virtual_key?.name ?? row.original.virtual_key_id), + cell: ({ row }) => , }, { id: "routing_rule", header: "Routing Rule", size: 170, - cell: ({ row }) => attributionCell(row.original.routing_rule?.name ?? row.original.routing_rule_id), + cell: ({ row }) => , }, { id: "team", header: "Team", size: 150, - cell: ({ row }) => attributionCell(row.original.team_name ?? row.original.team_id), + cell: ({ row }) => ( + + ), }, { id: "customer", header: "Customer", size: 150, - cell: ({ row }) => attributionCell(row.original.customer_name ?? row.original.customer_id), + cell: ({ row }) => ( + + ), }, { id: "user", header: "User", size: 150, - cell: ({ row }) => attributionCell(row.original.user_name ?? row.original.user_id), + cell: ({ row }) => , }, { id: "business_unit", header: "Business Unit", size: 150, - cell: ({ row }) => attributionCell(row.original.business_unit_name ?? row.original.business_unit_id), + cell: ({ row }) => ( + + ), }, ]; @@ -420,20 +486,20 @@ export const createColumns = ( const actionsColumn: ColumnDef[] = hasDeleteAccess ? [ - { - id: "actions", - header: "", - size: 56, - cell: ({ row }) => { - const log = row.original; - return ( -
- -
- ); - }, + { + id: "actions", + header: "", + size: 56, + cell: ({ row }) => { + const log = row.original; + return ( +
+ +
+ ); }, - ] + }, + ] : []; return [...baseColumns, ...attributionColumns, ...metadataColumns, ...actionsColumn]; diff --git a/ui/app/workspace/logs/views/logsHeaderView.tsx b/ui/app/workspace/logs/views/logsHeaderView.tsx index b0316dcd7b1..54244b5b2aa 100644 --- a/ui/app/workspace/logs/views/logsHeaderView.tsx +++ b/ui/app/workspace/logs/views/logsHeaderView.tsx @@ -5,8 +5,10 @@ import { DateTimePickerWithRange } from "@/components/ui/datePickerWithRange"; import { Input } from "@/components/ui/input"; import { Popover, PopoverContent, PopoverTrigger } from "@/components/ui/popover"; import { useTimezonePreference } from "@/lib/hooks/useTimezonePreference"; -import { getErrorMessage, useRecalculateLogCostsMutation } from "@/lib/store"; -import type { LogFilters as LogFiltersType } from "@/lib/types/logs"; +import { getErrorMessage } from "@/lib/store"; +import { getActiveTempToken } from "@/lib/store/apis/tempToken"; +import type { LogFilters as LogFiltersType, RecalculateCostProgress, RecalculateCostResponse } from "@/lib/types/logs"; +import { getApiBaseUrl } from "@/lib/utils/port"; import { getRangeForPeriod, TIME_PERIODS } from "@/lib/utils/timeRange"; import { Calculator, MoreVertical, Radio, RefreshCw, Search } from "lucide-react"; import { useCallback, useEffect, useRef, useState } from "react"; @@ -50,7 +52,6 @@ export function LogsHeaderView({ const [localSearch, setLocalSearch] = useState(filters.content_search || ""); const searchTimeoutRef = useRef(undefined); const filtersRef = useRef(filters); - const [recalculateCosts] = useRecalculateLogCostsMutation(); const [timezone, setTimezone] = useTimezonePreference(); @@ -77,19 +78,37 @@ export function LogsHeaderView({ }, []); const handleRecalculateCosts = useCallback(async () => { + setOpenMoreActionsPopover(false); + const toastId = "logs-recalculate-costs"; + const recalculatePromise = recalculateCostsWithProgress(filters, (progress) => { + const total = progress.total_matched || 0; + const processed = Math.min(progress.processed, total || progress.processed); + toast.loading("Recalculating log costs...", { + id: toastId, + description: + total > 0 + ? `${processed}/${total} checked, ${progress.updated} updated, ${progress.skipped} skipped` + : "Finding logs with missing costs", + }); + }); + + toast.promise(recalculatePromise, { + id: toastId, + loading: "Recalculating log costs...", + success: (response) => ({ + message: `Recalculated costs for ${response.updated} logs`, + description: `${response.updated} logs updated, ${response.skipped} logs skipped, ${response.remaining} logs remaining`, + duration: 5000, + }), + error: (err) => getErrorMessage(err), + }); + try { - const response = await recalculateCosts({ filters }).unwrap(); + await recalculatePromise; await fetchLogs(); await fetchStats(); - setOpenMoreActionsPopover(false); - toast.success(`Recalculated costs for ${response.updated} logs`, { - description: `${response.updated} logs updated, ${response.skipped} logs skipped, ${response.remaining} logs remaining`, - duration: 5000, - }); - } catch (err) { - toast.error(getErrorMessage(err)); - } - }, [filters, recalculateCosts, fetchLogs, fetchStats]); + } catch {} + }, [filters, fetchLogs, fetchStats]); const handleSearchChange = useCallback( (value: string) => { @@ -190,4 +209,107 @@ export function LogsHeaderView({ />
); +} + +async function recalculateCostsWithProgress( + filters: LogFiltersType, + onProgress: (progress: RecalculateCostProgress) => void, +): Promise { + const headers: Record = { + Accept: "text/event-stream", + "Content-Type": "application/json", + }; + const tempToken = getActiveTempToken(); + if (tempToken) { + headers["X-Bifrost-Temp-Token"] = tempToken; + } + + const response = await fetch(`${getApiBaseUrl()}/logs/recalculate-cost`, { + method: "POST", + credentials: "include", + headers, + body: JSON.stringify({ filters }), + }); + + if (!response.ok) { + throw await readRecalculateCostError(response); + } + if (!response.body) { + throw new Error("Recalculate cost stream is unavailable"); + } + + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + let buffer = ""; + let finalResult: RecalculateCostResponse | undefined; + + while (true) { + const { value, done } = await reader.read(); + if (done) break; + buffer += decoder.decode(value, { stream: true }); + const events = buffer.split("\n\n"); + buffer = events.pop() || ""; + for (const eventBlock of events) { + const parsed = parseSSEEvent(eventBlock); + if (!parsed || parsed.data === "[DONE]") continue; + if (parsed.event === "error") { + throw parseRecalculateCostStreamError(parsed.data); + } + if (parsed.event === "progress") { + onProgress(JSON.parse(parsed.data) as RecalculateCostProgress); + continue; + } + if (parsed.event === "done") { + finalResult = JSON.parse(parsed.data) as RecalculateCostResponse; + } + } + } + + buffer += decoder.decode(); + if (buffer.trim()) { + const parsed = parseSSEEvent(buffer); + if (parsed?.event === "error") { + throw parseRecalculateCostStreamError(parsed.data); + } + if (parsed?.event === "done") { + finalResult = JSON.parse(parsed.data) as RecalculateCostResponse; + } + } + + if (!finalResult) { + throw new Error("Recalculate cost stream ended before a final result was received"); + } + return finalResult; +} + +function parseSSEEvent(block: string): { event: string; data: string } | undefined { + let event = "message"; + const data: string[] = []; + for (const rawLine of block.split("\n")) { + const line = rawLine.trimEnd(); + if (line.startsWith("event: ")) { + event = line.slice(7); + } else if (line.startsWith("data: ")) { + data.push(line.slice(6)); + } + } + if (data.length === 0) return undefined; + return { event, data: data.join("\n") }; +} + +async function readRecalculateCostError(response: Response): Promise { + try { + return parseRecalculateCostStreamError(await response.text()); + } catch { + return new Error(`Failed to recalculate costs (${response.status})`); + } +} + +function parseRecalculateCostStreamError(data: string): Error { + try { + const parsed = JSON.parse(data) as { error?: { message?: string }; message?: string }; + return new Error(parsed.error?.message || parsed.message || "Failed to recalculate costs"); + } catch { + return new Error(data || "Failed to recalculate costs"); + } } \ No newline at end of file diff --git a/ui/app/workspace/logs/views/logsVolumeChart.tsx b/ui/app/workspace/logs/views/logsVolumeChart.tsx index 2138c368130..126ab6ae387 100644 --- a/ui/app/workspace/logs/views/logsVolumeChart.tsx +++ b/ui/app/workspace/logs/views/logsVolumeChart.tsx @@ -158,6 +158,13 @@ function CustomTooltip({ active, payload }: CustomTooltipProps) { {data.error.toLocaleString()}
+
+ + + Cancelled + + {(data.cancelled ?? 0).toLocaleString()} +
); @@ -218,6 +225,7 @@ export function LogsVolumeChart({ // If too many buckets, just return the original data without filling const result = (data.buckets || []).map((bucket, index) => ({ ...bucket, + cancelled: bucket.cancelled ?? 0, index, formattedTime: formatTimestamp(bucket.timestamp, data.bucket_size_seconds), })); @@ -229,6 +237,7 @@ export function LogsVolumeChart({ count: 0, success: 0, error: 0, + cancelled: 0, index: 1, formattedTime: formatTimestamp(nextTimestamp, data.bucket_size_seconds), }); @@ -245,6 +254,7 @@ export function LogsVolumeChart({ count: 0, success: 0, error: 0, + cancelled: 0, index: idx, formattedTime: formatTimestamp(timestamp, data.bucket_size_seconds), }); @@ -260,6 +270,7 @@ export function LogsVolumeChart({ if (bucketIndex >= 0 && bucketIndex < filledBuckets.length) { filledBuckets[bucketIndex] = { ...bucket, + cancelled: bucket.cancelled ?? 0, index: bucketIndex, formattedTime: formatTimestamp(bucket.timestamp, data.bucket_size_seconds), }; @@ -274,6 +285,7 @@ export function LogsVolumeChart({ count: 0, success: 0, error: 0, + cancelled: 0, index: 1, formattedTime: formatTimestamp(nextTimestamp, data.bucket_size_seconds), }); @@ -374,6 +386,10 @@ export function LogsVolumeChart({ Error + + + Cancelled +
)} {isZoomed && onResetZoom && ( @@ -441,6 +457,16 @@ export function LogsVolumeChart({ fill="#ef4444" barSize={30} fillOpacity={0.7} + radius={[0, 0, 0, 0]} + cursor="pointer" + onClick={(data) => handleBarClick(data?.payload as LogVolumeDataPoint | undefined)} + /> + handleBarClick(data?.payload as LogVolumeDataPoint | undefined)} @@ -459,4 +485,4 @@ export function LogsVolumeChart({ ); -} \ No newline at end of file +} diff --git a/ui/app/workspace/mcp-registry/page.tsx b/ui/app/workspace/mcp-registry/page.tsx index 0565d383532..b268f80a617 100644 --- a/ui/app/workspace/mcp-registry/page.tsx +++ b/ui/app/workspace/mcp-registry/page.tsx @@ -2,21 +2,84 @@ import FullPageLoader from "@/components/fullPageLoader"; import { useToast } from "@/hooks/use-toast"; import { useDebouncedValue } from "@/hooks/useDebounce"; import { getErrorMessage, useGetMCPClientsQuery } from "@/lib/store"; -import { useEffect, useState } from "react"; +import { parseAsArrayOf, parseAsBoolean, parseAsInteger, parseAsString, useQueryStates } from "nuqs"; +import { useCallback, useEffect, useMemo } from "react"; +import { MCPClientsFilterSidebar, type MCPClientFilters } from "./views/mcpClientsFilterSidebar"; import MCPClientsTable from "./views/mcpClientsTable"; const POLLING_INTERVAL = 5000; const PAGE_SIZE = 25; +// A boolean facet is only applied when exactly one option is selected; zero or +// both selections mean "no filter". Values are the underlying column strings. +function resolveBooleanFacet(selected: string[]): boolean | undefined { + if (selected.length !== 1) return undefined; + return selected[0] === "true"; +} + export default function MCPServersPage() { - const [search, setSearch] = useState(""); - const [offset, setOffset] = useState(0); - const debouncedSearch = useDebouncedValue(search, 300); + const [urlState, setUrlState] = useQueryStates( + { + search: parseAsString.withDefault(""), + server: parseAsString.withDefault(""), + connection_types: parseAsArrayOf(parseAsString).withDefault([]), + auth_types: parseAsArrayOf(parseAsString).withDefault([]), + states: parseAsArrayOf(parseAsString).withDefault([]), + code_mode: parseAsArrayOf(parseAsString).withDefault([]), + status: parseAsArrayOf(parseAsString).withDefault([]), + only_all_vks: parseAsBoolean.withDefault(false), + virtual_keys: parseAsArrayOf(parseAsString).withDefault([]), + offset: parseAsInteger.withDefault(0), + }, + { history: "replace" }, + ); + const debouncedSearch = useDebouncedValue(urlState.search, 300); - // Reset to first page when search changes - useEffect(() => { - setOffset(0); - }, [debouncedSearch]); + const filters: MCPClientFilters = useMemo( + () => ({ + connection_types: urlState.connection_types, + auth_types: urlState.auth_types, + states: urlState.states, + code_mode: urlState.code_mode, + status: urlState.status, + only_all_vks: urlState.only_all_vks, + virtual_keys: urlState.virtual_keys, + }), + [ + urlState.connection_types, + urlState.auth_types, + urlState.states, + urlState.code_mode, + urlState.status, + urlState.only_all_vks, + urlState.virtual_keys, + ], + ); + + const setFilters = useCallback( + (newFilters: MCPClientFilters) => { + void setUrlState({ + connection_types: newFilters.connection_types, + auth_types: newFilters.auth_types, + states: newFilters.states, + code_mode: newFilters.code_mode, + status: newFilters.status, + only_all_vks: newFilters.only_all_vks, + virtual_keys: newFilters.virtual_keys, + offset: 0, + }); + }, + [setUrlState], + ); + + const filtersActive = + filters.connection_types.length > 0 || + filters.auth_types.length > 0 || + filters.states.length > 0 || + filters.code_mode.length > 0 || + filters.status.length > 0 || + filters.only_all_vks || + filters.virtual_keys.length > 0; const { data: mcpClientsData, @@ -26,8 +89,16 @@ export default function MCPServersPage() { } = useGetMCPClientsQuery( { limit: PAGE_SIZE, - offset, + offset: urlState.offset, search: debouncedSearch || undefined, + server: urlState.server || undefined, + connection_type: filters.connection_types.length > 0 ? filters.connection_types.join(",") : undefined, + auth_type: filters.auth_types.length > 0 ? filters.auth_types.join(",") : undefined, + state: filters.states.length > 0 ? filters.states.join(",") : undefined, + virtual_keys: filters.virtual_keys.length > 0 ? filters.virtual_keys.join(",") : undefined, + all_virtual_keys: filters.only_all_vks || undefined, + code_mode: resolveBooleanFacet(filters.code_mode), + disabled: resolveBooleanFacet(filters.status), }, { pollingInterval: POLLING_INTERVAL, @@ -39,9 +110,9 @@ export default function MCPServersPage() { // Snap offset back when total shrinks past current page (e.g. delete last item on last page) useEffect(() => { - if (!mcpClientsData || offset < totalCount) return; - setOffset(totalCount === 0 ? 0 : Math.floor((totalCount - 1) / PAGE_SIZE) * PAGE_SIZE); - }, [totalCount, offset]); + if (!mcpClientsData || urlState.offset < totalCount) return; + void setUrlState({ offset: totalCount === 0 ? 0 : Math.floor((totalCount - 1) / PAGE_SIZE) * PAGE_SIZE }); + }, [totalCount, urlState.offset, mcpClientsData, setUrlState]); const { toast } = useToast(); @@ -61,19 +132,41 @@ export default function MCPServersPage() { return ; } + const handleSearchChange = (value: string) => void setUrlState({ search: value, offset: 0 }); + const handleServerFilterClear = () => void setUrlState({ server: null, offset: 0 }); + const handleOffsetChange = (offset: number) => void setUrlState({ offset }, { history: "push" }); + + const table = ( + + ); + + // Onboarding empty state: no servers at all and no active filters/search. + // Render full-width without the filter sidebar (the table renders the CTA). + if (totalCount === 0 && !filtersActive && !debouncedSearch && !urlState.server) { + return
{table}
; + } + return ( -
- +
+
+ +
+
{table}
+
+
); -} \ No newline at end of file +} diff --git a/ui/app/workspace/mcp-registry/views/mcpClientSheet.tsx b/ui/app/workspace/mcp-registry/views/mcpClientSheet.tsx index a775f9b5762..7f68572bf22 100644 --- a/ui/app/workspace/mcp-registry/views/mcpClientSheet.tsx +++ b/ui/app/workspace/mcp-registry/views/mcpClientSheet.tsx @@ -60,6 +60,28 @@ function toolSyncIntervalToMinutes(v: number | undefined | null): number { return n; } +/** API sends tool_execution_timeout as a Go duration string e.g. "30s". Normalize to whole seconds for form. */ +function toolExecutionTimeoutToSeconds(v: string | number | undefined | null): number { + if (v === undefined || v === null || v === "") return 0; + if (typeof v === "number") return v; + // Parse Go duration string: "30s", "1m30s", "2h", etc. + let total = 0; + const re = /(\d+(?:\.\d+)?)(ns|us|µs|ms|s|m|h)/g; + let match; + while ((match = re.exec(v)) !== null) { + const n = parseFloat(match[1]); + switch (match[2]) { + case "ns": total += n / 1e9; break; + case "us": case "µs": total += n / 1e6; break; + case "ms": total += n / 1e3; break; + case "s": total += n; break; + case "m": total += n * 60; break; + case "h": total += n * 3600; break; + } + } + return Math.ceil(total); +} + export default function MCPClientSheet({ mcpClient, onClose, @@ -191,6 +213,7 @@ export default function MCPClientSheet({ tools_to_auto_execute: mcpClient.config.tools_to_auto_execute || [], tool_pricing: mcpClient.config.tool_pricing || {}, tool_sync_interval: toolSyncIntervalToMinutes(mcpClient.config.tool_sync_interval), + tool_execution_timeout: toolExecutionTimeoutToSeconds(mcpClient.config.tool_execution_timeout), allowed_extra_headers: mcpClient.config.allowed_extra_headers || [], oauth_config: supportsOAuthCredentialUpdate ? { client_id: mcpClient.config.oauth_client_id, client_secret: mcpClient.config.oauth_client_secret } @@ -219,6 +242,7 @@ export default function MCPClientSheet({ tools_to_auto_execute: mcpClient.config.tools_to_auto_execute || [], tool_pricing: mcpClient.config.tool_pricing || {}, tool_sync_interval: toolSyncIntervalToMinutes(mcpClient.config.tool_sync_interval), + tool_execution_timeout: toolExecutionTimeoutToSeconds(mcpClient.config.tool_execution_timeout), allowed_extra_headers: mcpClient.config.allowed_extra_headers || [], oauth_config: supportsOAuthCredentialUpdate ? { client_id: mcpClient.config.oauth_client_id, client_secret: mcpClient.config.oauth_client_secret } @@ -282,6 +306,7 @@ export default function MCPClientSheet({ tools_to_auto_execute: data.tools_to_auto_execute, tool_pricing: data.tool_pricing, tool_sync_interval: data.tool_sync_interval ?? 0, + tool_execution_timeout: data.tool_execution_timeout ?? 0, allowed_extra_headers: data.allowed_extra_headers, oauth_config: shouldRotateOAuthCredentials ? { @@ -767,6 +792,58 @@ export default function MCPClientSheet({ ); }} /> + { + const isUsingGlobal = field.value === undefined || field.value === null || field.value === 0; + return ( + +
+
+
+ Tool Execution Timeout (seconds) +
+ + + + + + +

+ Override the global tool execution timeout for this server. Leave empty or set to 0 to use + the global setting. +

+
+
+
+
+
{isUsingGlobal &&

Using global setting

}
+
+ + { + if (e.target.value === "") { + field.onChange(undefined); + return; + } + const n = Number(e.target.value); + if (!Number.isInteger(n)) return; + field.onChange(n); + }} + min="0" + step="1" + data-testid="mcp-tool-execution-timeout" + /> + +
+ ); + }} + /> void; +} + +// --------------------------------------------------------------------------- +// MCPClientsFilterSidebar – orchestrator +// --------------------------------------------------------------------------- + +export function MCPClientsFilterSidebar({ filters, onFiltersChange }: SidebarProps) { + const [collapsed, setCollapsed] = useState(false); + + useEffect(() => { + if (typeof window === "undefined") return; + const stored = window.localStorage.getItem(COLLAPSE_STORAGE_KEY); + if (stored === "true") setCollapsed(true); + }, []); + + const toggleCollapsed = useCallback(() => { + setCollapsed((prev) => { + const next = !prev; + if (typeof window !== "undefined") { + window.localStorage.setItem(COLLAPSE_STORAGE_KEY, String(next)); + } + return next; + }); + }, []); + + const activeFilterCount = useMemo(() => { + return ( + filters.connection_types.length + + filters.auth_types.length + + filters.states.length + + (filters.code_mode.length === 1 ? 1 : 0) + + (filters.status.length === 1 ? 1 : 0) + + filters.virtual_keys.length + + (filters.only_all_vks ? 1 : 0) + ); + }, [filters]); + + const handleReset = useCallback(() => { + onFiltersChange(EMPTY_FILTERS); + }, [onFiltersChange]); + + if (collapsed) { + return ( + + ); + } + + return ( +
+
+ Filters +
+ {activeFilterCount > 0 && ( + + )} + +
+
+ + +
+ onFiltersChange({ ...filters, connection_types })} + testIdPrefix="mcp-clients-filter-connection-type" + /> + onFiltersChange({ ...filters, auth_types })} + testIdPrefix="mcp-clients-filter-auth-type" + /> + onFiltersChange({ ...filters, states })} + testIdPrefix="mcp-clients-filter-state" + /> + onFiltersChange({ ...filters, code_mode })} + testIdPrefix="mcp-clients-filter-code-mode" + /> + onFiltersChange({ ...filters, status })} + testIdPrefix="mcp-clients-filter-status" + /> + +
+
+
+ ); +} + +// --------------------------------------------------------------------------- +// Shared primitives +// --------------------------------------------------------------------------- + +function useAutoFocusOnOpen(isOpen: boolean) { + const ref = useRef(null); + // Skip the initial mount so focus isn't stolen when the section starts open + // from URL state; only focus on an explicit open action by the user. + const mounted = useRef(false); + useEffect(() => { + if (!mounted.current) { + mounted.current = true; + return; + } + if (isOpen) ref.current?.focus({ preventScroll: true }); + }, [isOpen]); + return ref; +} + +function FilterSection({ + title, + children, + defaultOpen = false, + onOpenChange, + testId, +}: { + title: string; + children: React.ReactNode; + defaultOpen?: boolean; + onOpenChange?: (open: boolean) => void; + testId?: string; +}) { + const [open, setOpen] = useState(defaultOpen); + + useEffect(() => { + if (defaultOpen) setOpen(true); + }, [defaultOpen]); + + const handleOpenChange = (next: boolean) => { + setOpen(next); + onOpenChange?.(next); + }; + + return ( + + + + {title} + + +
{children}
+
+
+ ); +} + +function CheckboxFilterItem({ + label, + checked, + onCheckedChange, + testId, +}: { + label: string; + checked: boolean; + onCheckedChange: (checked: boolean) => void; + testId?: string; +}) { + return ( + + ); +} + +function CheckboxFilterSection({ + title, + options, + selected, + defaultOpen = false, + onChange, + testIdPrefix, +}: { + title: string; + options: FilterOption[]; + selected: string[]; + defaultOpen?: boolean; + onChange: (selected: string[]) => void; + testIdPrefix?: string; +}) { + const hasActive = selected.length > 0; + + const toggle = (value: string) => { + if (selected.includes(value)) { + onChange(selected.filter((v) => v !== value)); + } else { + onChange([...selected, value]); + } + }; + + return ( + + {options.map((option) => ( + toggle(option.value)} + testId={testIdPrefix ? `${testIdPrefix}-checkbox-${option.value}` : undefined} + /> + ))} + + ); +} + +// SearchableCheckboxList – checkbox rows with a search input. Client-side label +// filtering is applied on top of the (debounced) onSearch callback so the caller +// can fetch server-side results. Mirrors the logs filter sidebar pattern. +function SearchableCheckboxList({ + items, + pinnedItems = [], + isSelected, + onToggle, + placeholder = "Search...", + inputRef, + testIdPrefix, + onSearch, + fetching, +}: { + items: { key: string; label: string }[]; + // Always-visible rows rendered before the searchable list and immune to the + // text filter (e.g. a static "All" option). Share isSelected/onToggle. + pinnedItems?: { key: string; label: string }[]; + isSelected: (key: string) => boolean; + onToggle: (key: string) => void; + placeholder?: string; + inputRef?: Ref; + testIdPrefix?: string; + onSearch?: (query: string) => void; + fetching?: boolean; +}) { + const [query, setQuery] = useState(""); + const normalized = query.trim().toLowerCase(); + const filtered = normalized ? items.filter((item) => item.label.toLowerCase().includes(normalized)) : items; + + useEffect(() => { + if (!onSearch) return; + const timer = setTimeout(() => onSearch(query.trim()), 300); + return () => clearTimeout(timer); + }, [query, onSearch]); + + return ( + <> + {pinnedItems.map((item) => ( + onToggle(item.key)} + testId={testIdPrefix ? `${testIdPrefix}-checkbox-${item.key}` : undefined} + /> + ))} +
+ {fetching ? ( + + ) : ( + + )} + setQuery(e.target.value)} + placeholder={placeholder} + className="h-8 border-0 pl-8 text-xs" + data-testid={testIdPrefix ? `${testIdPrefix}-search` : undefined} + /> +
+ {filtered.map((item) => ( + onToggle(item.key)} + testId={testIdPrefix ? `${testIdPrefix}-checkbox-${item.key}` : undefined} + /> + ))} + {filtered.length === 0 &&
No results
} + + ); +} + +// --------------------------------------------------------------------------- +// VKAccessFilterSection – a single checkbox list whose pinned first option is +// "All virtual keys" (allow_on_all_virtual_keys); the rest are individual VKs +// resolved via server-side search. They OR together server-side: a client +// matches if it is open to all VKs OR assigned to one of the selected VKs. +// --------------------------------------------------------------------------- + +// Reserved key for the pinned "All virtual keys" row — namespaced so it can +// never collide with a real virtual key id. +const ALL_VKS_KEY = "__all_virtual_keys__"; + +function VKAccessFilterSection({ filters, onFiltersChange }: SidebarProps) { + const hasActive = filters.only_all_vks || filters.virtual_keys.length > 0; + const [opened, setOpened] = useState(hasActive); + const [searchQuery, setSearchQuery] = useState(""); + const searchInputRef = useAutoFocusOnOpen(opened); + + const { data, isFetching } = useGetVirtualKeysQuery( + { limit: VK_PAGE_SIZE, offset: 0, search: searchQuery || undefined }, + { skip: !opened && !hasActive }, + ); + const virtualKeys = data?.virtual_keys || []; + + const isSelected = (key: string) => (key === ALL_VKS_KEY ? filters.only_all_vks : filters.virtual_keys.includes(key)); + + const toggle = (key: string) => { + if (key === ALL_VKS_KEY) { + onFiltersChange({ ...filters, only_all_vks: !filters.only_all_vks }); + return; + } + const current = filters.virtual_keys; + const next = current.includes(key) ? current.filter((v) => v !== key) : [...current, key]; + onFiltersChange({ ...filters, virtual_keys: next }); + }; + + return ( + + ({ key: vk.id, label: vk.name }))} + isSelected={isSelected} + onToggle={toggle} + onSearch={setSearchQuery} + fetching={isFetching} + testIdPrefix="mcp-clients-filter-vk" + /> + + ); +} diff --git a/ui/app/workspace/mcp-registry/views/mcpClientsTable.tsx b/ui/app/workspace/mcp-registry/views/mcpClientsTable.tsx index 05beb9e76e5..b2553543f22 100644 --- a/ui/app/workspace/mcp-registry/views/mcpClientsTable.tsx +++ b/ui/app/workspace/mcp-registry/views/mcpClientsTable.tsx @@ -22,7 +22,7 @@ import { getErrorMessage, useDeleteMCPClientMutation, useReconnectMCPClientMutat import { MCPClient } from "@/lib/types/mcp"; import { RbacOperation, RbacResource, useRbac } from "@enterprise/lib"; import { Link } from "@tanstack/react-router"; -import { Box, ChevronLeft, ChevronRight, Loader2, MoreHorizontal, PencilIcon, Plus, RefreshCcw, Search, Trash2 } from "lucide-react"; +import { Box, ChevronLeft, ChevronRight, Loader2, MoreHorizontal, PencilIcon, Plus, RefreshCcw, Search, Trash2, X } from "lucide-react"; import { useEffect, useMemo, useState } from "react"; import MCPClientSheet from "./mcpClientSheet"; import { MCPServersEmptyState } from "./mcpServersEmptyState"; @@ -125,7 +125,11 @@ interface MCPClientsTableProps { refetch?: () => void; search: string; debouncedSearch: string; + server: string; + /** Whether any sidebar facet filter (connection/auth/code-mode/status) is active. */ + filtersActive?: boolean; onSearchChange: (value: string) => void; + onServerFilterClear: () => void; offset: number; limit: number; onOffsetChange: (offset: number) => void; @@ -137,7 +141,10 @@ export default function MCPClientsTable({ refetch, search, debouncedSearch, + server, + filtersActive = false, onSearchChange, + onServerFilterClear, offset, limit, onOffsetChange, @@ -286,7 +293,7 @@ export default function MCPClientsTable({ } }; - const hasActiveFilters = debouncedSearch; + const hasActiveFilters = Boolean(debouncedSearch) || Boolean(server) || filtersActive; // True empty state: no servers at all (not just filtered to zero) if (totalCount === 0 && !hasActiveFilters) { @@ -372,23 +379,35 @@ export default function MCPClientsTable({ data-testid="mcp-clients-search-input" />
+ {server && ( + + )}
- +
- Name - Connection Type - Auth Type - Auth Scope - Code Mode - VK Access - Enabled Tools - Auto-execute Tools - State - Status + Name + Connection Type + Auth Type + Auth Scope + Code Mode + VK Access + Enabled Tools + Auto-execute Tools + State + Status @@ -419,7 +438,11 @@ export default function MCPClientsTable({ : 0; return ( - {c.config.name} + +
+ {c.config.name} +
+
{getConnectionTypeDisplay(c.config.connection_type)} diff --git a/ui/app/workspace/mcp-sessions/page.tsx b/ui/app/workspace/mcp-sessions/page.tsx index 48a25dc0bd8..721595ecddc 100644 --- a/ui/app/workspace/mcp-sessions/page.tsx +++ b/ui/app/workspace/mcp-sessions/page.tsx @@ -18,12 +18,14 @@ export default function MCPSessionsPage() { status: parseAsArrayOf(parseAsString).withDefault([]), auth_mode: parseAsArrayOf(parseAsString).withDefault([]), mcp_client_id: parseAsArrayOf(parseAsString).withDefault([]), + identity: parseAsString.withDefault(""), offset: parseAsInteger.withDefault(0), }, { history: "push" }, ); const debouncedSearch = useDebouncedValue(urlState.q, 300); + const normalizedIdentity = urlState.identity.trim(); const { data, isLoading, isFetching, isError, error } = useGetMCPSessionsQuery({ q: debouncedSearch || undefined, @@ -31,6 +33,7 @@ export default function MCPSessionsPage() { status: urlState.status.length ? (urlState.status as MCPSessionStatus[]) : undefined, auth_mode: urlState.auth_mode.length ? (urlState.auth_mode as AuthMode[]) : undefined, mcp_client_id: urlState.mcp_client_id.length ? urlState.mcp_client_id : undefined, + identity: normalizedIdentity || undefined, limit: PAGE_SIZE, offset: urlState.offset, }); @@ -46,9 +49,7 @@ export default function MCPSessionsPage() { }); }, [totalCount, urlState.offset, data, setUrlState]); - if (isLoading) { - return ; - } + if (isLoading) return ; if (isError) { return ( @@ -65,17 +66,18 @@ export default function MCPSessionsPage() { urlState.kind.length > 0 || urlState.status.length > 0 || urlState.auth_mode.length > 0 || - urlState.mcp_client_id.length > 0; + urlState.mcp_client_id.length > 0 || + !!normalizedIdentity; const handleSearchChange = (value: string) => setUrlState({ q: value || null, offset: 0 }); const handleKindChange = (value: string[]) => setUrlState({ kind: value.length ? value : null, offset: 0 }); const handleStatusChange = (value: string[]) => setUrlState({ status: value.length ? value : null, offset: 0 }); const handleAuthModeChange = (value: string[]) => setUrlState({ auth_mode: value.length ? value : null, offset: 0 }); const handleOffsetChange = (offset: number) => setUrlState({ offset }); - const handleClearFilters = () => setUrlState({ q: null, kind: null, status: null, auth_mode: null, mcp_client_id: null, offset: 0 }); + const handleClearFilters = () => setUrlState({ q: null, kind: null, status: null, auth_mode: null, mcp_client_id: null, identity: null, offset: 0 }); return ( -
+
+
!open && setPendingDelete(null)}> @@ -148,7 +148,7 @@ export default function SessionsTable({ -
+

MCP Auth Sessions

@@ -157,134 +157,147 @@ export default function SessionsTable({

- +
+ +
-
-
- - - MCP server - - - - - - - - - - - - - Created - - - - - {sessions.length === 0 ? ( +
+
+
+ - - {hasActiveFilters ? ( -
No sessions match these filters.
- ) : ( - - No sessions yet. Sessions appear here when an inference request or MCP gateway call triggers per-user authentication - (OAuth or header submission). - - )} -
+ MCP server + + + + + + + + + + + + + Created +
- ) : ( - sessions.map((row) => ( - - {row.mcp_client?.name || row.mcp_client?.client_id || "-"} - - - - - - - - - - -
- {formatAccessExpiry(row)} - {row.last_refreshed_at && refreshed {formatRelativePast(row.last_refreshed_at)}} -
-
- {formatRelativePast(row.created_at)} - - handleReauth(row)} - onRevoke={() => setPendingDelete(row)} - /> +
+ + {sessions.length === 0 ? ( + + + {hasActiveFilters ? ( +
No sessions match these filters.
+ ) : ( + + No sessions yet. Sessions appear here when an inference request or MCP gateway call triggers per-user authentication + (OAuth or header submission). + + )}
- )) - )} -
-
-
+ ) : ( + sessions.map((row) => ( + + {row.mcp_client?.name || row.mcp_client?.client_id || "-"} + + + + + + + + + + +
+ {formatAccessExpiry(row)} + {row.last_refreshed_at && refreshed {formatRelativePast(row.last_refreshed_at)}} +
+
+ {formatRelativePast(row.created_at)} + + handleReauth(row)} + onRevoke={() => setPendingDelete(row)} + /> + +
+ )) + )} + + +
- {totalCount > 0 && ( -
-

- Showing {offset + 1}-{Math.min(offset + limit, totalCount)} of {totalCount} -

-
- - + {totalCount > 0 && ( +
+
+ {(offset + 1).toLocaleString()}-{Math.min(offset + limit, totalCount).toLocaleString()} of {totalCount.toLocaleString()}{" "} + entries +
+ +
+ + +
+ Page + {Math.floor(offset / limit) + 1} + of {Math.ceil(totalCount / limit)} +
+ + +
-
- )} + )} +
); } diff --git a/ui/app/workspace/model-catalog/views/overviewTab.tsx b/ui/app/workspace/model-catalog/views/overviewTab.tsx index 83eba0fc20e..dcabd21e142 100644 --- a/ui/app/workspace/model-catalog/views/overviewTab.tsx +++ b/ui/app/workspace/model-catalog/views/overviewTab.tsx @@ -1,7 +1,13 @@ import FullPageLoader from "@/components/fullPageLoader"; import { ProviderNames } from "@/lib/constants/logs"; -import { useGetModelsQuery, useGetProvidersQuery, useLazyGetLogsModelHistogramQuery, useLazyGetLogsStatsQuery } from "@/lib/store"; -import { KnownProvider } from "@/lib/types/config"; +import { + useGetModelsQuery, + useGetProvidersQuery, + useLazyGetLogsModelHistogramQuery, + useLazyGetLogsStatsQuery, + useLazyGetProviderKeysQuery, +} from "@/lib/store"; +import { KnownProvider, ModelProviderKey } from "@/lib/types/config"; import { LogStats } from "@/lib/types/logs"; import { useEffect, useMemo, useState } from "react"; import { ModelCatalogEmptyState } from "./modelCatalogEmptyState"; @@ -11,6 +17,37 @@ interface OverviewTabProps { hasAccess: boolean; } +function buildAliasDisplayMap(keys: ModelProviderKey[]) { + const displayByValue = new Map(); + + for (const key of keys) { + for (const [aliasName, rawConfig] of Object.entries(key.aliases ?? {})) { + const config = rawConfig as unknown; + const modelValues = + typeof config === "string" + ? [config] + : typeof config === "object" && config !== null + ? [(config as { model_id?: string }).model_id, (config as { model_name?: string }).model_name].filter( + (value): value is string => Boolean(value), + ) + : []; + + for (const modelValue of modelValues) { + const aliases = displayByValue.get(modelValue) ?? []; + if (!aliases.includes(aliasName)) { + displayByValue.set(modelValue, [...aliases, aliasName]); + } + } + } + } + + return displayByValue; +} + +function getDisplayModels(models: string[], displayByValue: Map) { + return models.flatMap((model) => displayByValue.get(model) ?? [model]); +} + export default function OverviewTab({ hasAccess }: OverviewTabProps) { const [providerFilter, setProviderFilter] = useState(""); const [statsMap, setStatsMap] = useState>(new Map()); @@ -28,6 +65,7 @@ export default function OverviewTab({ hasAccess }: OverviewTabProps) { const [triggerGlobalStats, { data: globalStats }] = useLazyGetLogsStatsQuery(); const [triggerStats] = useLazyGetLogsStatsQuery(); const [triggerModelHistogram] = useLazyGetLogsModelHistogramQuery(); + const [triggerProviderKeys] = useLazyGetProviderKeysQuery(); useEffect(() => { if (!hasAccess) return; @@ -79,12 +117,19 @@ export default function OverviewTab({ hasAccess }: OverviewTabProps) { const monthAgo = new Date(Date.now() - 30 * 24 * 60 * 60 * 1000).toISOString(); Promise.all( - providers.map((p) => - triggerModelHistogram({ filters: { providers: [p.name], start_time: monthAgo, end_time: now } }) - .unwrap() - .then((data): [string, string[]] => [p.name, data.models ?? []]) - .catch((): [string, string[]] => [p.name, []]), - ), + providers.map(async (p): Promise<[string, string[]]> => { + const [models, keys] = await Promise.all([ + triggerModelHistogram({ filters: { providers: [p.name], start_time: monthAgo, end_time: now } }) + .unwrap() + .then((data) => data.models ?? []) + .catch(() => []), + triggerProviderKeys(p.name) + .unwrap() + .catch(() => []), + ]); + + return [p.name, getDisplayModels(models, buildAliasDisplayMap(keys))]; + }), ).then((results) => { if (!cancelled) { setModelsUsedMap(new Map(results)); @@ -94,7 +139,7 @@ export default function OverviewTab({ hasAccess }: OverviewTabProps) { return () => { cancelled = true; }; - }, [providers, triggerModelHistogram]); + }, [providers, triggerModelHistogram, triggerProviderKeys]); const rows: ModelCatalogRow[] = useMemo(() => { if (!providers) return []; diff --git a/ui/app/workspace/model-limits/views/modelLimitsTable.tsx b/ui/app/workspace/model-limits/views/modelLimitsTable.tsx index 576d2e7cd6a..f743c54d4f8 100644 --- a/ui/app/workspace/model-limits/views/modelLimitsTable.tsx +++ b/ui/app/workspace/model-limits/views/modelLimitsTable.tsx @@ -410,7 +410,7 @@ export default function ModelLimitsTable({ ) : ( - — + - )} diff --git a/ui/app/workspace/oauth-grants/layout.tsx b/ui/app/workspace/oauth-grants/layout.tsx new file mode 100644 index 00000000000..f8e855022b8 --- /dev/null +++ b/ui/app/workspace/oauth-grants/layout.tsx @@ -0,0 +1,6 @@ +import { createFileRoute } from "@tanstack/react-router"; +import OAuthGrantsPage from "./page"; + +export const Route = createFileRoute("/workspace/oauth-grants")({ + component: OAuthGrantsPage, +}); diff --git a/ui/app/workspace/oauth-grants/page.tsx b/ui/app/workspace/oauth-grants/page.tsx new file mode 100644 index 00000000000..04ad057887d --- /dev/null +++ b/ui/app/workspace/oauth-grants/page.tsx @@ -0,0 +1,125 @@ +import { useDebouncedValue } from "@/hooks/useDebounce"; +import { getErrorMessage, useGetOAuth2GrantsQuery, useRevokeOAuth2GrantMutation } from "@/lib/store"; +import type { OAuth2GrantRow } from "@/lib/store/apis/oauth2SessionsApi"; +import { Loader2 } from "lucide-react"; +import { parseAsArrayOf, parseAsInteger, parseAsString, useQueryStates } from "nuqs"; +import { useEffect, useState } from "react"; +import { toast } from "sonner"; +import GrantsFilterBar from "./views/grantsFilterBar"; +import GrantsTable from "./views/grantsTable"; +import RevokeGrantDialog from "./views/revokeGrantDialog"; + +const PAGE_SIZE = 50; + +export default function OAuthGrantsPage() { + const [urlState, setUrlState] = useQueryStates( + { + q: parseAsString.withDefault(""), + bf_mode: parseAsArrayOf(parseAsString).withDefault([]), + offset: parseAsInteger.withDefault(0), + }, + { history: "push" }, + ); + + const debouncedSearch = useDebouncedValue(urlState.q, 300); + + const { data, isLoading, isFetching, isError, error } = useGetOAuth2GrantsQuery({ + q: debouncedSearch || undefined, + bf_mode: urlState.bf_mode.length ? urlState.bf_mode : undefined, + limit: PAGE_SIZE, + offset: urlState.offset, + }); + const [revokeGrant, { isLoading: revoking }] = useRevokeOAuth2GrantMutation(); + + const [pendingDelete, setPendingDelete] = useState(null); + const [pendingActionRowId, setPendingActionRowId] = useState(null); + + const page = data?.sessions ?? []; + const totalCount = data?.total_count ?? 0; + const hasActiveFilters = !!urlState.q || urlState.bf_mode.length > 0; + + // Snap the offset back into range when the total shrinks past the current + // page (e.g. a revoke removes the last row on the last page). Without this + // the page goes blank with the paginator and clear-filters affordances both + // hidden. Mirrors the MCP sessions page. + useEffect(() => { + if (!data || urlState.offset < totalCount) return; + setUrlState({ + offset: totalCount === 0 ? 0 : Math.floor((totalCount - 1) / PAGE_SIZE) * PAGE_SIZE, + }); + }, [totalCount, urlState.offset, data, setUrlState]); + + const handleSearchChange = (value: string) => setUrlState({ q: value || null, offset: 0 }); + const handleModeChange = (value: string[]) => setUrlState({ bf_mode: value.length ? value : null, offset: 0 }); + const handleOffsetChange = (offset: number) => setUrlState({ offset }); + const clearFilters = () => setUrlState({ q: null, bf_mode: null, offset: 0 }); + + const confirmRevoke = async () => { + if (!pendingDelete) return; + const row = pendingDelete; + setPendingDelete(null); + setPendingActionRowId(row.id); + try { + await revokeGrant(row.id).unwrap(); + toast.success("Grant revoked"); + } catch (err) { + toast.error("Failed to revoke grant", { description: getErrorMessage(err) }); + } finally { + setPendingActionRowId(null); + } + }; + + return ( +
+ !open && setPendingDelete(null)} + onConfirm={confirmRevoke} + /> + +
+
+

OAuth Grants

+

+ Active downstream OAuth grants issued to MCP clients that connected + via the OAuth consent flow. +

+
+
+ +
+ +
+ + {isLoading ? ( +
+ +
+ ) : isError ? ( +
+ Failed to load OAuth grants: {getErrorMessage(error)} +
+ ) : ( + + )} +
+ ); +} diff --git a/ui/app/workspace/oauth-grants/views/grantActions.tsx b/ui/app/workspace/oauth-grants/views/grantActions.tsx new file mode 100644 index 00000000000..09903b3dfaf --- /dev/null +++ b/ui/app/workspace/oauth-grants/views/grantActions.tsx @@ -0,0 +1,61 @@ +// Per-row actions menu for an OAuth grant: a dropdown exposing "View auth +// sessions" (deep-link into MCP sessions pre-filtered to this grant's exact +// identity) and a destructive "Revoke" action. The revoke confirmation itself +// is owned by the page via onRevoke. + +import { Button } from "@/components/ui/button"; +import { + DropdownMenu, + DropdownMenuContent, + DropdownMenuItem, + DropdownMenuTrigger, +} from "@/components/ui/dropdownMenu"; +import type { OAuth2GrantRow } from "@/lib/store/apis/oauth2SessionsApi"; +import { Link } from "@tanstack/react-router"; +import { ExternalLink, Loader2, MoreHorizontal, Trash2 } from "lucide-react"; + +interface GrantActionsProps { + row: OAuth2GrantRow; + revoking: boolean; + isPendingRow: boolean; + onRevoke: () => void; +} + +export default function GrantActions({ row, revoking, isPendingRow, onRevoke }: GrantActionsProps) { + const busy = revoking; + // Link to Auth Sessions pre-filtered to this grant's exact identity: the + // mode plus the identity filter, which exact-matches bf_sub against the + // session's user_id / virtual key id / session id, so the user lands on + // just this identity's sessions. + const authSessionsUrl = `/workspace/mcp-sessions?auth_mode=${row.bf_mode}&identity=${encodeURIComponent(row.bf_sub)}`; + + return ( + + + + + + {(row.bf_mode === "user" || row.bf_mode === "vk" || row.bf_mode === "session") && ( + + + + View auth sessions + + + )} + { e.preventDefault(); onRevoke(); }} + > + + Revoke + + + + ); +} diff --git a/ui/app/workspace/oauth-grants/views/grantsFilterBar.tsx b/ui/app/workspace/oauth-grants/views/grantsFilterBar.tsx new file mode 100644 index 00000000000..84e647758a1 --- /dev/null +++ b/ui/app/workspace/oauth-grants/views/grantsFilterBar.tsx @@ -0,0 +1,65 @@ +// Filter row above the grants table: free-text search (client / identity) and +// a multi-select on the grant's identity mode (user / vk / session), plus a +// clear-filters affordance shown only when something is active. + +import { Button } from "@/components/ui/button"; +import { ComboboxSelect } from "@/components/ui/combobox"; +import { Input } from "@/components/ui/input"; +import { Fingerprint, KeyRound, Search, UserRound, X } from "lucide-react"; + +const MODE_OPTIONS = [ + { label: "User", value: "user", icon: }, + { label: "Virtual key", value: "vk", icon: }, + { label: "Session", value: "session", icon: }, +]; + +interface GrantsFilterBarProps { + search: string; + onSearchChange: (value: string) => void; + modeFilter: string[]; + onModeChange: (value: string[]) => void; + hasActiveFilters: boolean; + onClearFilters: () => void; +} + +export default function GrantsFilterBar({ + search, + onSearchChange, + modeFilter, + onModeChange, + hasActiveFilters, + onClearFilters, +}: GrantsFilterBarProps) { + return ( +
+
+ + onSearchChange(e.target.value)} + className="pl-9" + /> +
+ + {hasActiveFilters && ( + + )} +
+ ); +} diff --git a/ui/app/workspace/oauth-grants/views/grantsTable.tsx b/ui/app/workspace/oauth-grants/views/grantsTable.tsx new file mode 100644 index 00000000000..127251e72f4 --- /dev/null +++ b/ui/app/workspace/oauth-grants/views/grantsTable.tsx @@ -0,0 +1,258 @@ +// Results table for OAuth grants: one row per active downstream grant, with the +// bound identity, an approximate access-token expiry, created/last-used relative +// times, and a per-row actions menu. Owns the empty state and pagination; the +// page passes in the current page slice plus filter/revoke state. + +import { Button } from "@/components/ui/button"; +import { + Table, + TableBody, + TableCell, + TableHead, + TableHeader, + TableRow, +} from "@/components/ui/table"; +import { Tooltip, TooltipContent, TooltipProvider, TooltipTrigger } from "@/components/ui/tooltip"; +import { PIN_SHADOW_RIGHT } from "@/components/table/columnPinning"; +import type { OAuth2GrantRow } from "@/lib/store/apis/oauth2SessionsApi"; +import { + ChevronLeft, + ChevronRight, + Fingerprint, + Info, + KeyRound, + UserRound, +} from "lucide-react"; +import GrantActions from "./grantActions"; + +interface GrantsTableProps { + rows: OAuth2GrantRow[]; + totalCount: number; + offset: number; + pageSize: number; + onOffsetChange: (offset: number) => void; + isFetching: boolean; + hasActiveFilters: boolean; + revoking: boolean; + pendingActionRowId: string | null; + onRevoke: (row: OAuth2GrantRow) => void; +} + +export default function GrantsTable({ + rows, + totalCount, + offset, + pageSize, + onOffsetChange, + isFetching, + hasActiveFilters, + revoking, + pendingActionRowId, + onRevoke, +}: GrantsTableProps) { + return ( +
+
+ + + + Client + + + + + + + Created + + + + + + + + {rows.length === 0 ? ( + + + {hasActiveFilters ? ( +
+ No grants match these filters. +
+ ) : ( + + )} +
+
+ ) : ( + rows.map((row) => ( + + + {row.client_name || row.client_id} + + + + + + + + + {formatRelativePast(row.created_at)} + + + {formatRelativePast(row.last_used_at || row.created_at)} + + + onRevoke(row)} + /> + + + )) + )} +
+
+
+ + {totalCount > 0 && ( +
+
+ {(offset + 1).toLocaleString()}-{Math.min(offset + pageSize, totalCount).toLocaleString()} of {totalCount.toLocaleString()} entries +
+ +
+ + +
+ Page + {Math.floor(offset / pageSize) + 1} + of {Math.ceil(totalCount / pageSize)} +
+ + +
+
+ )} +
+ ); +} + +function BindingCell({ row }: { row: OAuth2GrantRow }) { + const display = row.bf_sub_display || row.bf_sub; + if (row.bf_mode === "user") { + return ( + + + {display} + + ); + } + if (row.bf_mode === "vk") { + return ( + + + {display} + + ); + } + return ( + + + {display} + + ); +} + +function AccessTokenExpiry({ row }: { row: OAuth2GrantRow }) { + // Access token TTL is 10 min (600s default). Access tokens are stateless JWTs + // not stored server-side, so we approximate expiry from the grant's last + // activity (last_used_at, falling back to created_at). Anchoring to created_at + // alone would read as expired for any grant that has silently refreshed. + const baseMs = new Date(row.last_used_at ?? row.created_at).getTime(); + if (!Number.isFinite(baseMs)) { + return Unknown; + } + const expiryMs = baseMs + 600_000; // 10 min default + const diffMs = expiryMs - Date.now(); + if (diffMs < 0) { + return Refreshes on next use; + } + const mins = Math.ceil(diffMs / 60_000); + return in {mins} min; +} + +function HeaderWithTooltip({ label, tooltip }: { label: string; tooltip: string }) { + return ( + + + + + {label} + + + + {tooltip} + + + ); +} + +function EmptyGrantsState() { + return ( +
+

+ No grants yet. Grants appear here when an MCP client connects via the OAuth + consent flow. (Authentication Mode needs to be set to "oauth" or "both" for grants to be issued.) +

+
+ ); +} + +function formatRelativePast(iso: string): string { + try { + const ts = new Date(iso).getTime(); + if (!Number.isFinite(ts)) return iso; + const diffMs = Date.now() - ts; + if (diffMs < 0) return "just now"; + const mins = Math.floor(diffMs / 60_000); + if (mins < 1) return "just now"; + if (mins < 60) return `${mins}m ago`; + const hrs = Math.floor(mins / 60); + if (hrs < 24) return `${hrs}h ago`; + const days = Math.floor(hrs / 24); + return `${days}d ago`; + } catch { + return iso; + } +} diff --git a/ui/app/workspace/oauth-grants/views/revokeGrantDialog.tsx b/ui/app/workspace/oauth-grants/views/revokeGrantDialog.tsx new file mode 100644 index 00000000000..5de22477df2 --- /dev/null +++ b/ui/app/workspace/oauth-grants/views/revokeGrantDialog.tsx @@ -0,0 +1,51 @@ +// Confirmation dialog for revoking an OAuth grant. Open/confirm are driven by +// the page; the copy explains that the refresh token stops rotating immediately +// while the current short-lived access token keeps working until it expires. + +import { + AlertDialog, + AlertDialogAction, + AlertDialogCancel, + AlertDialogContent, + AlertDialogDescription, + AlertDialogFooter, + AlertDialogHeader, + AlertDialogTitle, +} from "@/components/ui/alertDialog"; + +interface RevokeGrantDialogProps { + open: boolean; + onOpenChange: (open: boolean) => void; + onConfirm: () => void; +} + +export default function RevokeGrantDialog({ open, onOpenChange, onConfirm }: RevokeGrantDialogProps) { + return ( + + + + Revoke this OAuth grant? + + The refresh token for this grant stops rotating right away, so the + MCP client can no longer renew its access. Its current access token + is a short-lived JWT that keeps working on the{" "} + /mcp{" "} + endpoint until it expires (default 10 minutes), after which the client + is fully cut off and must reconnect via the OAuth consent flow. + + + + + Cancel + + + Revoke + + + + + ); +} diff --git a/ui/app/workspace/pii-redactor/layout.tsx b/ui/app/workspace/pii-redactor/layout.tsx deleted file mode 100644 index 93ebd25fab6..00000000000 --- a/ui/app/workspace/pii-redactor/layout.tsx +++ /dev/null @@ -1,17 +0,0 @@ -import { createFileRoute, Outlet, useChildMatches } from "@tanstack/react-router"; -import { NoPermissionView } from "@/components/noPermissionView"; -import { RbacOperation, RbacResource, useRbac } from "@enterprise/lib"; -import PiiRedactorPage from "./page"; - -function RouteComponent() { - const hasPiiRedactorAccess = useRbac(RbacResource.PIIRedactor, RbacOperation.View); - const childMatches = useChildMatches(); - if (!hasPiiRedactorAccess) { - return ; - } - return childMatches.length === 0 ? : ; -} - -export const Route = createFileRoute("/workspace/pii-redactor")({ - component: RouteComponent, -}); \ No newline at end of file diff --git a/ui/app/workspace/pii-redactor/page.tsx b/ui/app/workspace/pii-redactor/page.tsx deleted file mode 100644 index a62f010a06b..00000000000 --- a/ui/app/workspace/pii-redactor/page.tsx +++ /dev/null @@ -1,9 +0,0 @@ -import PiiRedactorRulesView from "@enterprise/components/pii-redactor/piiRedactorRulesView"; - -export default function PiiRedactorPage() { - return ( -
- -
- ); -} \ No newline at end of file diff --git a/ui/app/workspace/pii-redactor/providers/layout.tsx b/ui/app/workspace/pii-redactor/providers/layout.tsx deleted file mode 100644 index 8b2cc72e5c3..00000000000 --- a/ui/app/workspace/pii-redactor/providers/layout.tsx +++ /dev/null @@ -1,6 +0,0 @@ -import { createFileRoute } from "@tanstack/react-router"; -import PiiRedactorProvidersPage from "./page"; - -export const Route = createFileRoute("/workspace/pii-redactor/providers")({ - component: PiiRedactorProvidersPage, -}); \ No newline at end of file diff --git a/ui/app/workspace/pii-redactor/providers/page.tsx b/ui/app/workspace/pii-redactor/providers/page.tsx deleted file mode 100644 index ea231659f8c..00000000000 --- a/ui/app/workspace/pii-redactor/providers/page.tsx +++ /dev/null @@ -1,9 +0,0 @@ -import PiiRedactorProviderView from "@enterprise/components/pii-redactor/piiRedactorProviderView"; - -export default function PiiRedactorProvidersPage() { - return ( -
- -
- ); -} \ No newline at end of file diff --git a/ui/app/workspace/pii-redactor/rules/layout.tsx b/ui/app/workspace/pii-redactor/rules/layout.tsx deleted file mode 100644 index 05b182ccf12..00000000000 --- a/ui/app/workspace/pii-redactor/rules/layout.tsx +++ /dev/null @@ -1,6 +0,0 @@ -import { createFileRoute } from "@tanstack/react-router"; -import PiiRedactorRulesPage from "./page"; - -export const Route = createFileRoute("/workspace/pii-redactor/rules")({ - component: PiiRedactorRulesPage, -}); \ No newline at end of file diff --git a/ui/app/workspace/pii-redactor/rules/page.tsx b/ui/app/workspace/pii-redactor/rules/page.tsx deleted file mode 100644 index f79d4159376..00000000000 --- a/ui/app/workspace/pii-redactor/rules/page.tsx +++ /dev/null @@ -1,9 +0,0 @@ -import PiiRedactorRulesView from "@enterprise/components/pii-redactor/piiRedactorRulesView"; - -export default function PiiRedactorRulesPage() { - return ( -
- -
- ); -} \ No newline at end of file diff --git a/ui/app/workspace/providers/dialogs/addNewCustomProviderSheet.tsx b/ui/app/workspace/providers/dialogs/addNewCustomProviderSheet.tsx index 4f4b669804b..c258b75e881 100644 --- a/ui/app/workspace/providers/dialogs/addNewCustomProviderSheet.tsx +++ b/ui/app/workspace/providers/dialogs/addNewCustomProviderSheet.tsx @@ -55,6 +55,10 @@ export function AddCustomProviderSheetContent({ show = true, onClose, onSave }: chat_completion_stream: true, responses: true, responses_stream: true, + responses_retrieve: true, + responses_delete: true, + responses_cancel: true, + responses_input_items: true, embedding: true, speech: true, speech_stream: true, diff --git a/ui/app/workspace/providers/dialogs/providerConfigSheet.tsx b/ui/app/workspace/providers/dialogs/providerConfigSheet.tsx index 828eae7252d..f587795578b 100644 --- a/ui/app/workspace/providers/dialogs/providerConfigSheet.tsx +++ b/ui/app/workspace/providers/dialogs/providerConfigSheet.tsx @@ -21,7 +21,7 @@ interface Props { provider: ModelProvider; } -const ANTHROPIC_FAMILY_PROVIDERS = ["anthropic", "vertex", "bedrock", "azure"]; +const ANTHROPIC_FAMILY_PROVIDERS = ["anthropic", "vertex", "bedrock", "bedrock_mantle", "azure"]; const availableTabs = (hasCustomProviderConfig: boolean, hasGovernanceAccess: boolean, isOpenAI: boolean, isAnthropicFamily: boolean) => { const tabs = []; diff --git a/ui/app/workspace/providers/fragments/allowedRequestsFields.tsx b/ui/app/workspace/providers/fragments/allowedRequestsFields.tsx index 5fbd46302f0..9a44aa7242e 100644 --- a/ui/app/workspace/providers/fragments/allowedRequestsFields.tsx +++ b/ui/app/workspace/providers/fragments/allowedRequestsFields.tsx @@ -70,6 +70,10 @@ const RequestTypes: Array<{ key: RequestType; label: string }> = [ { key: "chat_completion_stream", label: "Chat Completion Stream" }, { key: "responses", label: "Responses" }, { key: "responses_stream", label: "Responses Stream" }, + { key: "responses_retrieve", label: "Responses Retrieve" }, + { key: "responses_delete", label: "Responses Delete" }, + { key: "responses_cancel", label: "Responses Cancel" }, + { key: "responses_input_items", label: "Responses Input Items" }, { key: "embedding", label: "Embedding" }, { key: "speech", label: "Speech" }, { key: "speech_stream", label: "Speech Stream" }, @@ -83,6 +87,15 @@ const RequestTypes: Array<{ key: RequestType; label: string }> = [ { key: "count_tokens", label: "Count Tokens" }, ]; +// Path overrides replace the default path verbatim; these request paths embed the +// response ID, so an override can never produce a valid URL for them. +const PathOverrideUnsupported = new Set([ + "responses_retrieve", + "responses_delete", + "responses_cancel", + "responses_input_items", +]); + export function AllowedRequestsFields({ control, namePrefix = "allowed_requests", @@ -122,7 +135,7 @@ export function AllowedRequestsFields({
{/* Settings icon for path override - only show when enabled */} - {allowedField.value && !isDisabled && !isPathOverrideDisabled && !disabled && ( + {allowedField.value && !isDisabled && !isPathOverrideDisabled && !disabled && !PathOverrideUnsupported.has(requestType.key) && ( ("iam_role"); + // Auth type state for Bedrock Mantle: 'iam_role', 'explicit', or 'api_key' + const [bedrockMantleAuthType, setBedrockMantleAuthType] = useState<"iam_role" | "explicit" | "api_key">("iam_role"); + // Auth type state for Vertex: 'service_account', 'service_account_json', or 'api_key' const [vertexAuthType, setVertexAuthType] = useState<"service_account" | "service_account_json" | "api_key">("service_account"); @@ -130,6 +134,27 @@ export function ApiKeyFormFragment({ control, providerName, baseProviderType, fo } }, [isBedrock, form]); + useEffect(() => { + if (form.formState.isDirty) return; + if (isBedrockMantle) { + const accessKey = form.getValues("key.bedrock_mantle_key_config.access_key"); + const secretKey = form.getValues("key.bedrock_mantle_key_config.secret_key"); + const apiKey = form.getValues("key.value"); + const hasExplicitCreds = accessKey?.value || accessKey?.ref || secretKey?.value || secretKey?.ref; + const hasApiKey = apiKey?.value || apiKey?.ref; + let detected: "iam_role" | "explicit" | "api_key" = "iam_role"; + if (hasExplicitCreds) { + detected = "explicit"; + } else if (hasApiKey) { + detected = "api_key"; + } + setBedrockMantleAuthType(detected); + form.setValue("key.bedrock_mantle_key_config._auth_type", detected); + } + // form.formState.defaultValues is a dependency so detection re-runs when ProviderKeyForm + // repopulates an existing key via form.reset(...) after mount, not only on first render. + }, [isBedrockMantle, form, form.formState.defaultValues]); + return (
@@ -201,7 +226,7 @@ export function ApiKeyFormFragment({ control, providerName, baseProviderType, fo />
{/* Hide API Key field for providers with dedicated auth tabs */} - {!isAzure && !isBedrock && !isVertex && ( + {!isAzure && !isBedrock && !isBedrockMantle && !isVertex && ( }
)} + + {isBedrockMantle && ( +
+ +
+ Authentication Method + { + setBedrockMantleAuthType(v as "iam_role" | "explicit" | "api_key"); + form.setValue("key.bedrock_mantle_key_config._auth_type", v, { shouldDirty: true, shouldValidate: true }); + if (v === "iam_role") { + // Clear explicit credentials and API key when switching to IAM Role + form.setValue("key.bedrock_mantle_key_config.access_key", undefined, { shouldDirty: true }); + form.setValue("key.bedrock_mantle_key_config.secret_key", undefined, { shouldDirty: true }); + form.setValue("key.bedrock_mantle_key_config.session_token", undefined, { shouldDirty: true }); + form.setValue("key.value", undefined, { shouldDirty: true }); + } else if (v === "explicit") { + // Clear API key when switching to Explicit Credentials + form.setValue("key.value", undefined, { shouldDirty: true }); + } else if (v === "api_key") { + // Clear AWS credentials and assume-role fields when switching to API Key + form.setValue("key.bedrock_mantle_key_config.access_key", undefined, { shouldDirty: true }); + form.setValue("key.bedrock_mantle_key_config.secret_key", undefined, { shouldDirty: true }); + form.setValue("key.bedrock_mantle_key_config.session_token", undefined, { shouldDirty: true }); + form.setValue("key.bedrock_mantle_key_config.role_arn", undefined, { shouldDirty: true }); + form.setValue("key.bedrock_mantle_key_config.external_id", undefined, { shouldDirty: true }); + form.setValue("key.bedrock_mantle_key_config.session_name", undefined, { shouldDirty: true }); + } + }} + > + + + IAM Role (Inherited) + + + Explicit Credentials + + + API Key + + + + {bedrockMantleAuthType === "iam_role" && ( +

Uses IAM roles attached to your environment (EC2, Lambda, ECS, EKS).

+ )} + {bedrockMantleAuthType === "api_key" && ( +

Uses a Bedrock Mantle API key sent as a Bearer token.

+ )} +
+ + {bedrockMantleAuthType === "explicit" && ( + <> + ( + + Access Key (Required) + + + + + + )} + /> + ( + + Secret Key (Required) + + + + + + )} + /> + ( + + Session Token (Optional) + + + + + + )} + /> + + )} + + {bedrockMantleAuthType === "api_key" && ( + ( + + API Key + + + + + + )} + /> + )} + + ( + + Region (Required) + + + + + + )} + /> + + {bedrockMantleAuthType !== "api_key" && ( + <> + ( + + Assume Role ARN (Optional) + + Assume an IAM role before requests. Works with both explicit credentials and inherited IAM (EC2, ECS, EKS). + + + + + + + )} + /> + ( + + External ID (Optional) + Required by the role's trust policy when using cross-account access. + + + + + + )} + /> + ( + + Session Name (Optional) + AssumeRole session name (defaults to bifrost-session). + + + + + + )} + /> + + )} +
+ )}
); } \ No newline at end of file diff --git a/ui/app/workspace/providers/fragments/apiStructureFormFragment.tsx b/ui/app/workspace/providers/fragments/apiStructureFormFragment.tsx index 7721c43eb8c..c63c5f8b20c 100644 --- a/ui/app/workspace/providers/fragments/apiStructureFormFragment.tsx +++ b/ui/app/workspace/providers/fragments/apiStructureFormFragment.tsx @@ -43,6 +43,10 @@ export function ApiStructureFormFragment({ provider }: Props) { chat_completion_stream: provider.custom_provider_config?.allowed_requests?.chat_completion_stream ?? true, responses: provider.custom_provider_config?.allowed_requests?.responses ?? true, responses_stream: provider.custom_provider_config?.allowed_requests?.responses_stream ?? true, + responses_retrieve: provider.custom_provider_config?.allowed_requests?.responses_retrieve ?? false, + responses_delete: provider.custom_provider_config?.allowed_requests?.responses_delete ?? false, + responses_cancel: provider.custom_provider_config?.allowed_requests?.responses_cancel ?? false, + responses_input_items: provider.custom_provider_config?.allowed_requests?.responses_input_items ?? false, embedding: provider.custom_provider_config?.allowed_requests?.embedding ?? true, speech: provider.custom_provider_config?.allowed_requests?.speech ?? true, speech_stream: provider.custom_provider_config?.allowed_requests?.speech_stream ?? true, diff --git a/ui/app/workspace/providers/fragments/betaHeadersFormFragment.tsx b/ui/app/workspace/providers/fragments/betaHeadersFormFragment.tsx index 8454a42edc6..2088f74fddb 100644 --- a/ui/app/workspace/providers/fragments/betaHeadersFormFragment.tsx +++ b/ui/app/workspace/providers/fragments/betaHeadersFormFragment.tsx @@ -22,87 +22,87 @@ const KNOWN_BETA_HEADERS = [ prefix: "computer-use-", label: "Computer Use", description: "Computer use client tool", - defaults: { anthropic: true, vertex: true, bedrock: true, azure: true }, + defaults: { anthropic: true, vertex: true, bedrock: true, bedrock_mantle: true, azure: true }, }, { prefix: "structured-outputs-", label: "Structured Outputs", description: "Strict tool validation and output_format", - defaults: { anthropic: true, vertex: false, bedrock: true, azure: true }, + defaults: { anthropic: true, vertex: false, bedrock: true, bedrock_mantle: true, azure: true }, }, { prefix: "advanced-tool-use-", label: "Advanced Tool Use", description: "defer_loading, input_examples, allowed_callers", - defaults: { anthropic: true, vertex: false, bedrock: false, azure: true }, + defaults: { anthropic: true, vertex: false, bedrock: false, bedrock_mantle: false, azure: true }, }, { prefix: "mcp-client-", label: "MCP Client", description: "MCP connector support", - defaults: { anthropic: true, vertex: false, bedrock: false, azure: true }, + defaults: { anthropic: true, vertex: false, bedrock: false, bedrock_mantle: false, azure: true }, }, { prefix: "prompt-caching-scope-", label: "Prompt Caching Scope", description: "Prompt caching scope control", - defaults: { anthropic: true, vertex: false, bedrock: false, azure: true }, + defaults: { anthropic: true, vertex: false, bedrock: false, bedrock_mantle: false, azure: true }, }, { prefix: "compact-", label: "Compaction", description: "Server-side context compaction", - defaults: { anthropic: true, vertex: true, bedrock: true, azure: true }, + defaults: { anthropic: true, vertex: true, bedrock: true, bedrock_mantle: true, azure: true }, }, { prefix: "context-management-", label: "Context Management", description: "Context editing (clear_tool_uses, clear_thinking)", - defaults: { anthropic: true, vertex: true, bedrock: true, azure: true }, + defaults: { anthropic: true, vertex: true, bedrock: true, bedrock_mantle: true, azure: true }, }, { prefix: "files-api-", label: "Files API", description: "Files API support", - defaults: { anthropic: true, vertex: false, bedrock: false, azure: true }, + defaults: { anthropic: true, vertex: false, bedrock: false, bedrock_mantle: false, azure: true }, }, { prefix: "interleaved-thinking-", label: "Interleaved Thinking", description: "Interleaved thinking between tool calls", - defaults: { anthropic: true, vertex: true, bedrock: true, azure: true }, + defaults: { anthropic: true, vertex: true, bedrock: true, bedrock_mantle: true, azure: true }, }, { prefix: "skills-", label: "Skills", description: "Agent Skills", - defaults: { anthropic: true, vertex: false, bedrock: false, azure: true }, + defaults: { anthropic: true, vertex: false, bedrock: false, bedrock_mantle: false, azure: true }, }, { prefix: "context-1m-", label: "Context 1M", description: "1M context window (beta for Sonnet 4.5/4)", - defaults: { anthropic: true, vertex: true, bedrock: true, azure: true }, + defaults: { anthropic: true, vertex: true, bedrock: true, bedrock_mantle: true, azure: true }, }, { prefix: "fast-mode-", label: "Fast Mode", description: "Fast mode (Opus 4.6 research preview)", - defaults: { anthropic: true, vertex: false, bedrock: false, azure: false }, + defaults: { anthropic: true, vertex: false, bedrock: false, bedrock_mantle: false, azure: false }, }, { prefix: "redact-thinking-", label: "Redact Thinking", description: "Redact thinking blocks in responses", - defaults: { anthropic: true, vertex: false, bedrock: false, azure: true }, + defaults: { anthropic: true, vertex: false, bedrock: false, bedrock_mantle: false, azure: true }, }, ] as const; const KNOWN_PREFIXES = new Set(KNOWN_BETA_HEADERS.map((h) => h.prefix)); -type ProviderKey = "anthropic" | "vertex" | "bedrock" | "azure"; +type ProviderKey = "anthropic" | "vertex" | "bedrock" | "bedrock_mantle" | "azure"; -const ANTHROPIC_FAMILY_PROVIDERS: ProviderKey[] = ["anthropic", "vertex", "bedrock", "azure"]; +const ANTHROPIC_FAMILY_PROVIDERS: ProviderKey[] = ["anthropic", "vertex", "bedrock", "bedrock_mantle", "azure"]; function getProviderKey(providerName: string): ProviderKey | null { const name = providerName.toLowerCase(); diff --git a/ui/app/workspace/providers/fragments/governanceFormFragment.tsx b/ui/app/workspace/providers/fragments/governanceFormFragment.tsx index bf85814af06..c26095ac8ab 100644 --- a/ui/app/workspace/providers/fragments/governanceFormFragment.tsx +++ b/ui/app/workspace/providers/fragments/governanceFormFragment.tsx @@ -5,6 +5,7 @@ import MultiBudgetLines, { BudgetLineEntry } from "@/components/ui/multibudgets" import NumberAndSelect from "@/components/ui/numberAndSelect"; import { DottedSeparator } from "@/components/ui/separator"; import { Switch } from "@/components/ui/switch"; +import { supportsCalendarAlignment } from "@/lib/constants/governance"; import { getErrorMessage, useDeleteProviderGovernanceMutation, @@ -87,6 +88,8 @@ export function GovernanceFormFragment({ provider }: GovernanceFormFragmentProps const watchedBudgets = form.watch("budgets"); const watchedCalendarAligned = form.watch("calendarAligned"); + const configuredBudgets = watchedBudgets.filter((b) => b.max_limit !== undefined && b.max_limit > 0); + const showCalendarAlignment = configuredBudgets.some((b) => supportsCalendarAlignment(b.reset_duration)); useEffect(() => { if (providerGovernance && !form.formState.isDirty) { @@ -100,9 +103,18 @@ export function GovernanceFormFragment({ provider }: GovernanceFormFragmentProps form.reset(governanceToFormValues(newProvGov)); }, [provider.name, form]); + // Drop a stale calendarAligned when no configured budget supports alignment, so the + // toggle never reappears pre-enabled if an alignable budget is added back later. + useEffect(() => { + if (!showCalendarAlignment && watchedCalendarAligned) { + form.setValue("calendarAligned", false, { shouldDirty: true }); + } + }, [showCalendarAlignment, watchedCalendarAligned, form]); + const onSubmit = async (data: FormData) => { try { const validBudgets = data.budgets.filter((b) => b.max_limit !== undefined && b.max_limit > 0); + const hasAlignableBudget = validBudgets.some((b) => supportsCalendarAlignment(b.reset_duration)); const hadBudgets = (providerGovernance?.budgets?.length ?? 0) > 0; const hadRateLimit = !!providerGovernance?.rate_limit; const hasRateLimit = data.tokenMaxLimit !== undefined || data.requestMaxLimit !== undefined; @@ -141,7 +153,7 @@ export function GovernanceFormFragment({ provider }: GovernanceFormFragmentProps provider: provider.name, data: { budgets: budgetsPayload, - ...(budgetsPayload !== undefined ? { calendar_aligned: data.calendarAligned } : {}), + ...(budgetsPayload !== undefined ? { calendar_aligned: hasAlignableBudget && data.calendarAligned } : {}), rate_limit: rateLimitPayload, }, }).unwrap(); @@ -177,8 +189,8 @@ export function GovernanceFormFragment({ provider }: GovernanceFormFragmentProps onChange={(lines) => form.setValue("budgets", lines, { shouldDirty: true })} /> - {/* Calendar Alignment — only shown when there are budgets */} - {watchedBudgets.length > 0 && ( + {/* Calendar Alignment — only shown when a budget uses a calendar-alignable period (day/week/month/year) */} + {showCalendarAlignment && (