Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions litellm/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -1837,6 +1837,9 @@ def set_global_gitlab_config(config: Dict[str, Any]) -> None:
from .llms.openrouter.responses.transformation import (
OpenRouterResponsesAPIConfig as OpenRouterResponsesAPIConfig,
)
from .llms.bedrock.responses.transformation import (
BedrockOpenAIResponsesConfig as BedrockOpenAIResponsesConfig,
)
from .llms.bedrock_mantle.responses.transformation import (
BedrockMantleResponsesAPIConfig as BedrockMantleResponsesAPIConfig,
)
Expand Down
5 changes: 5 additions & 0 deletions litellm/_lazy_imports_registry.py
Original file line number Diff line number Diff line change
Expand Up @@ -242,6 +242,7 @@
"PerplexityResponsesConfig",
"DatabricksResponsesAPIConfig",
"OpenRouterResponsesAPIConfig",
"BedrockOpenAIResponsesConfig",
"BedrockMantleResponsesAPIConfig",
"GoogleAIStudioInteractionsConfig",
"VertexAIInteractionsConfig",
Expand Down Expand Up @@ -900,6 +901,10 @@
"OpenAITextCompletionConfig",
),
"GroqChatConfig": (".llms.groq.chat.transformation", "GroqChatConfig"),
"BedrockOpenAIResponsesConfig": (
".llms.bedrock.responses.transformation",
"BedrockOpenAIResponsesConfig",
),
"BedrockMantleChatConfig": (
".llms.bedrock_mantle.chat.transformation",
"BedrockMantleChatConfig",
Expand Down
130 changes: 130 additions & 0 deletions litellm/llms/base_llm/responses/codex_compat.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,130 @@
"""Codex CLI wire-format quirks shared by the Responses API providers that need them.

Codex sends history item types that api.openai.com accepts but other Responses
backends reject with ``400 Invalid 'input': value did not match any expected
variant``. Both Amazon Bedrock endpoints reject them:

- ``bedrock-mantle.{region}.api.aws`` (verified against ``openai.gpt-5.6-sol``)
- ``bedrock-runtime.{region}.amazonaws.com/openai/v1`` (same, verified separately)

They are *history* items, so they only appear from the second turn of a session
onward -- a first-turn request succeeds and hides the problem entirely.

The normalizer is a pure transform that reports which types it rewrote; callers do
their own logging, so each provider keeps its own wording.
"""

import json
from collections.abc import Mapping
from typing import Final

from typing_extensions import ReadOnly, TypedDict

from litellm.types.llms.openai import ResponseInputParam

AGENT_MESSAGE_INPUT_ITEM_TYPE: Final = "agent_message"
CONTEXT_COMPACTION_INPUT_ITEM_TYPE: Final = "context_compaction"
LOCAL_SHELL_CALL_INPUT_ITEM_TYPE: Final = "local_shell_call"


class _RewrittenOutputTextBlock(TypedDict):
type: ReadOnly[str]
text: ReadOnly[str]


class _RewrittenAssistantMessageItem(TypedDict):
type: ReadOnly[str]
role: ReadOnly[str]
content: ReadOnly[tuple[_RewrittenOutputTextBlock, ...]]


class _RewrittenCompactionItem(TypedDict):
type: ReadOnly[str]
encrypted_content: ReadOnly[str]


class _RewrittenFunctionCallItem(TypedDict):
type: ReadOnly[str]
call_id: ReadOnly[str]
name: ReadOnly[str]
arguments: ReadOnly[str]


def _agent_message_text(item: "Mapping[str, object]") -> str:
content: Final = item.get("content")
if not isinstance(content, list):
return ""
return "".join(
str(block.get("text") or block.get("encrypted_content") or "") for block in content if isinstance(block, dict)
)


def _normalize_agent_message_item(item: "Mapping[str, object]") -> "_RewrittenAssistantMessageItem | None":
text: Final = _agent_message_text(item)
if not text:
return None
rewritten: Final[_RewrittenAssistantMessageItem] = {
"type": "message",
"role": "assistant",
"content": ({"type": "output_text", "text": text},),
}
return rewritten


def _normalize_context_compaction_item(item: "Mapping[str, object]") -> "_RewrittenCompactionItem | None":
encrypted_content: Final = item.get("encrypted_content")
if not isinstance(encrypted_content, str) or not encrypted_content:
return None
rewritten: Final[_RewrittenCompactionItem] = {"type": "compaction", "encrypted_content": encrypted_content}
return rewritten


def _normalize_local_shell_call_item(item: "Mapping[str, object]") -> "_RewrittenFunctionCallItem | None":
call_id: Final = item.get("call_id")
if not isinstance(call_id, str) or not call_id:
return None
action: Final = item.get("action")
rewritten: Final[_RewrittenFunctionCallItem] = {
"type": "function_call",
"call_id": call_id,
"name": "local_shell",
"arguments": json.dumps(action) if isinstance(action, dict) else "{}",
}
return rewritten


def _normalize_input_item(item: object) -> "tuple[object, str | None]":
"""Returns (normalized item, or None to drop it; original type when rewritten)."""
if not isinstance(item, dict):
return item, None
item_type: Final = item.get("type")
if item_type == AGENT_MESSAGE_INPUT_ITEM_TYPE:
return _normalize_agent_message_item(item), item_type
if item_type == CONTEXT_COMPACTION_INPUT_ITEM_TYPE:
return _normalize_context_compaction_item(item), item_type
if item_type == LOCAL_SHELL_CALL_INPUT_ITEM_TYPE:
return _normalize_local_shell_call_item(item), item_type
return item, None


def normalize_codex_input_items(
input: "str | ResponseInputParam",
) -> "tuple[str | ResponseInputParam, tuple[str, ...]]":
"""Rewrite the Codex history item types a Responses backend rejects.

``agent_message`` (Codex multi-agent traffic; its ``encrypted_content`` slot
carries the plaintext payload when the model never issued encrypted args)
becomes an assistant message, ``context_compaction`` becomes the ``compaction``
spelling these backends accept, and ``local_shell_call`` becomes the
``function_call`` its recorded ``function_call_output`` already pairs with.

Returns the normalized input and the sorted set of types that were rewritten,
so the caller can log in its own words. Non-list input is returned untouched.
"""
if not isinstance(input, list):
return input, ()
normalized: Final = tuple(_normalize_input_item(item) for item in input)
rewritten_types: Final = tuple(sorted(frozenset(item_type for _, item_type in normalized if item_type is not None)))
kept: Final = [i for i, _ in normalized if i is not None] # mutable-ok: downstream narrows on isinstance(list)
# Codex passthrough items sit outside the OpenAI input union.
return kept, rewritten_types # pyright: ignore[reportReturnType] # see above
22 changes: 22 additions & 0 deletions litellm/llms/bedrock/common_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -805,6 +805,28 @@ def _mantle_api_base_from_env() -> str | None:
return next((base[: -len(suffix)] for suffix in _MANTLE_OPENAI_BASE_SUFFIXES if base.endswith(suffix)), base)


def bedrock_supports_openai_responses(model: str | None, model_cost: Mapping[str, object]) -> bool:
"""Whether a Bedrock model is served by bedrock-runtime's OpenAI Responses surface.

Purely data-driven from the model's price-map capability signal -- ``/v1/responses``
in ``supported_endpoints`` -- and overridable via ``register_model`` and proxy
``model_info``, so onboarding a model is a JSON change, never a code change.
There is deliberately no model-name match: AWS exposes this surface per model,
not per family, and the two Bedrock endpoints do not agree with each other
(bedrock-runtime accepts Codex's ``additional_tools`` items where
bedrock-mantle rejects them), so a name-shaped gate would be wrong.
A model absent from ``model_cost`` has no signal and returns False, leaving the
chat-completions bridge in place exactly as before.
"""
if not model:
return False
candidates: Final = (model_cost.get(key) for key in (model, f"bedrock/{model}"))
return any(
isinstance(entry, Mapping) and "/v1/responses" in (entry.get("supported_endpoints") or ())
for entry in candidates
)


def build_mantle_messages_url(
api_base: str | None,
aws_bedrock_runtime_endpoint: str | None,
Expand Down
159 changes: 159 additions & 0 deletions litellm/llms/bedrock/responses/transformation.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,159 @@
"""Amazon Bedrock Runtime - native OpenAI Responses API.

AWS serves the OpenAI models on ``bedrock-runtime`` through an OpenAI-compatible
surface at ``https://bedrock-runtime.{region}.{dns_suffix}/openai/v1/responses``,
alongside Converse. Without this config the ``bedrock`` provider has no Responses
config at all, so ``/v1/responses`` falls back to the Chat Completions bridge and
the request is translated into Converse. A realistic Codex session does not
survive that translation: its ``function_call`` / ``function_call_output`` history
becomes Converse ``toolUse`` / ``toolResult`` blocks with no ``toolConfig``, and
Converse rejects the request outright with "The toolConfig field must be defined
when using toolUse and toolResult content blocks".

Payloads and SSE follow the OpenAI Responses spec, so this inherits
OpenAIResponsesAPIConfig and overrides only the endpoint URL, authentication, and
the Codex history-item normalization the endpoint requires.

Auth: Bearer token (litellm_params.api_key or the standard AWS_BEARER_TOKEN_BEDROCK)
when present; otherwise AWS SigV4 (service "bedrock") over the standard credential
chain, signed via BaseAWSLLM._sign_request once the body is final.

Model IDs: bedrock-runtime serves these models only through a cross-Region
inference profile, so the model is named ``us.openai.gpt-5.6-sol`` or
``global.openai.gpt-5.6-sol``; there is no in-Region form.
"""

from typing import Final

import httpx

import litellm
from litellm._logging import verbose_logger
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.base_llm.responses.codex_compat import normalize_codex_input_items
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
from litellm.llms.bedrock.common_utils import (
BedrockError,
bedrock_supports_openai_responses,
)
from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig
from litellm.secret_managers.main import get_secret_str
from litellm.types.llms.openai import ResponseInputParam
from litellm.types.router import GenericLiteLLMParams
from litellm.types.utils import LlmProviders

BEDROCK_RUNTIME_OPENAI_RESPONSES_PATH: Final = "/openai/v1/responses"


def resolve_bedrock_bearer_token(api_key: str | None) -> str | None:
return api_key or get_secret_str("AWS_BEARER_TOKEN_BEDROCK")


class BedrockOpenAIResponsesConfig(BaseAWSLLM, OpenAIResponsesAPIConfig):
"""Responses API config for the OpenAI models on the bedrock-runtime endpoint."""

@classmethod
def for_model(cls, model: str | None) -> "BedrockOpenAIResponsesConfig | None":
"""This config when ``model`` is served on the OpenAI Responses surface, else ``None``.

The capability decision lives here rather than in the shared dispatch so that
onboarding a model, or changing how the signal is read, stays inside the
Bedrock adapter. ``None`` leaves the caller's existing behaviour untouched --
chat-only Bedrock models keep the Chat Completions bridge.
"""
if not bedrock_supports_openai_responses(model, litellm.model_cost):
return None
return cls()

@property
def custom_llm_provider(self) -> LlmProviders:
return LlmProviders.BEDROCK

def get_error_class(
self, error_message: str, status_code: int, headers: dict[str, object] | httpx.Headers
) -> BaseLLMException:
# The OpenAI base builds a blank response, dropping x-amzn-RequestId.
return BedrockError(status_code=status_code, message=error_message, headers=headers)

def get_complete_url(
self,
api_base: str | None,
litellm_params: dict, # mutable-ok: signature fixed by the BaseResponsesAPIConfig override contract
) -> str:
region: Final = self._get_aws_region_name(optional_params=litellm_params, model=None)
override: Final = (
api_base
or litellm_params.get("aws_bedrock_runtime_endpoint")
or get_secret_str("AWS_BEDROCK_RUNTIME_ENDPOINT")
)
# Partition-aware: bedrock-runtime is amazonaws.com.cn in China, and other
# suffixes in GovCloud/ISO, so defer to the shared endpoint builder.
host: Final = (
override or self._select_default_endpoint_url(endpoint_type="runtime", aws_region_name=region)
).rstrip("/")
if host.endswith(BEDROCK_RUNTIME_OPENAI_RESPONSES_PATH):
return host
base: Final = next(
(host[: -len(suffix)] for suffix in ("/openai/v1", "/v1") if host.endswith(suffix)),
host,
)
return f"{base}{BEDROCK_RUNTIME_OPENAI_RESPONSES_PATH}"

def validate_environment(
self,
headers: dict, # mutable-ok: signature fixed by the BaseResponsesAPIConfig override contract
model: str,
litellm_params: GenericLiteLLMParams | None,
) -> dict: # mutable-ok: signature fixed by the BaseResponsesAPIConfig override contract
api_key: Final = litellm_params.api_key if litellm_params is not None else None
bearer: Final = resolve_bedrock_bearer_token(api_key)
if not bearer:
return headers
return {**headers, "Authorization": f"Bearer {bearer}"} # mutable-ok: dict return per the contract

def sign_request(
self,
headers: dict, # mutable-ok: signature fixed by the BaseResponsesAPIConfig override contract
optional_params: dict, # mutable-ok: same
request_data: dict, # mutable-ok: same
api_base: str,
api_key: str | None = None,
model: str | None = None,
stream: bool | None = None,
fake_stream: bool | None = None,
) -> "tuple[dict, bytes | None]": # mutable-ok: signature fixed by the override contract
if resolve_bedrock_bearer_token(api_key):
# Bedrock API keys are Bearer credentials; SigV4 on top would be wrong.
return headers, None
return self._sign_request(
service_name="bedrock",
headers=headers,
optional_params=optional_params,
request_data=request_data,
api_base=api_base,
model=model,
stream=stream,
fake_stream=fake_stream,
)

def transform_responses_api_request(
self,
model: str,
input: "str | ResponseInputParam",
response_api_optional_request_params: dict, # mutable-ok: signature fixed by the override contract
litellm_params: GenericLiteLLMParams,
headers: dict, # mutable-ok: same
) -> dict: # mutable-ok: same
normalized_input, rewritten_types = normalize_codex_input_items(input)
if rewritten_types:
verbose_logger.warning(
"Bedrock Runtime Responses API: rewrote Codex input item type(s) %s that the endpoint rejects.",
rewritten_types,
)
return super().transform_responses_api_request(
model=model,
input=normalized_input,
response_api_optional_request_params=response_api_optional_request_params,
litellm_params=litellm_params,
headers=headers,
)
Loading
Loading