Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 9 additions & 3 deletions litellm/llms/bedrock/chat/converse_transformation.py
Original file line number Diff line number Diff line change
Expand Up @@ -115,6 +115,12 @@
]


def _is_openai_frontier_model(model: str) -> bool:
"""OpenAI GPT-5.x / GPT-6 on Bedrock (``openai.gpt-5.6-sol``, ``openai.gpt-6-astra``): reasoning goes
through ``reasoning.effort``. ``openai.gpt-oss-*`` is excluded; it takes ``reasoning_effort`` as-is."""
return "openai.gpt-" in model and "gpt-oss" not in model


class AmazonConverseConfig(BaseConfig):
"""
Reference - https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_Converse.html
Expand Down Expand Up @@ -430,7 +436,7 @@ def _handle_reasoning_effort_parameter(self, model: str, reasoning_effort: str,
"""
if "gpt-oss" in model:
optional_params["reasoning_effort"] = reasoning_effort
elif "openai.gpt-5" in model:
elif _is_openai_frontier_model(model):
reasoning: Final[BedrockConverseGptReasoningEffortBlock] = {"effort": reasoning_effort}
optional_params["reasoning"] = reasoning
elif self._is_nova_2_model(model):
Expand Down Expand Up @@ -564,7 +570,7 @@ def get_supported_openai_params(self, model: str) -> list[str]:
# only anthropic and mistral support tool choice config. otherwise (E.g. cohere) will fail the call - https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_ToolChoice.html
supported_params.append("tool_choice")

if "gpt-oss" in model or "openai.gpt-5" in model or "openai.gpt-5" in base_model:
if "gpt-oss" in model or _is_openai_frontier_model(model) or _is_openai_frontier_model(base_model):
supported_params.append("reasoning_effort")
elif self._is_nova_2_model(model):
# Nova 2 models support reasoning_effort (transformed to reasoningConfig)
Expand Down Expand Up @@ -920,7 +926,7 @@ def map_openai_params(
optional_params["_parallel_tool_use_config"] = {
"tool_choice": {"type": "auto", "disable_parallel_tool_use": not value}
}
if param == "thinking" and "openai.gpt-5" not in model:
if param == "thinking" and not _is_openai_frontier_model(model):
if (
isinstance(value, dict)
and value.get("type") == "adaptive"
Expand Down
54 changes: 54 additions & 0 deletions litellm/model_prices_and_context_window_backup.json
Original file line number Diff line number Diff line change
Expand Up @@ -55196,6 +55196,60 @@
"supports_reasoning": true,
"supports_vision": true
},
"us.openai.gpt-6-astra": {
"input_cost_per_token": 1.1e-05,
"input_cost_per_token_above_272k_tokens": 2.2e-05,
"cache_creation_input_token_cost": 1.375e-05,
"cache_creation_input_token_cost_above_272k_tokens": 2.75e-05,
"cache_read_input_token_cost": 1.1e-06,
"cache_read_input_token_cost_above_272k_tokens": 2.2e-06,
"output_cost_per_token": 5.5e-05,
"output_cost_per_token_above_272k_tokens": 8.25e-05,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 1050000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"supports_prompt_caching": true,
"supports_vision": true
},
"global.openai.gpt-6-astra": {
"input_cost_per_token": 1e-05,
"input_cost_per_token_above_272k_tokens": 2e-05,
"cache_creation_input_token_cost": 1.25e-05,
"cache_creation_input_token_cost_above_272k_tokens": 2.5e-05,
"cache_read_input_token_cost": 1e-06,
"cache_read_input_token_cost_above_272k_tokens": 2e-06,
"output_cost_per_token": 5e-05,
"output_cost_per_token_above_272k_tokens": 7.5e-05,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 1050000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"supports_prompt_caching": true,
"supports_vision": true
},
"bedrock_mantle/openai.gpt-5.5": {
"input_cost_per_token": 5.5e-06,
"input_cost_per_token_above_272k_tokens": 1.1e-05,
Expand Down
54 changes: 54 additions & 0 deletions model_prices_and_context_window.json
Original file line number Diff line number Diff line change
Expand Up @@ -55196,6 +55196,60 @@
"supports_reasoning": true,
"supports_vision": true
},
"us.openai.gpt-6-astra": {
"input_cost_per_token": 1.1e-05,
"input_cost_per_token_above_272k_tokens": 2.2e-05,
"cache_creation_input_token_cost": 1.375e-05,
"cache_creation_input_token_cost_above_272k_tokens": 2.75e-05,
"cache_read_input_token_cost": 1.1e-06,
"cache_read_input_token_cost_above_272k_tokens": 2.2e-06,
"output_cost_per_token": 5.5e-05,
"output_cost_per_token_above_272k_tokens": 8.25e-05,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 1050000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"supports_prompt_caching": true,
"supports_vision": true
},
"global.openai.gpt-6-astra": {
"input_cost_per_token": 1e-05,
"input_cost_per_token_above_272k_tokens": 2e-05,
"cache_creation_input_token_cost": 1.25e-05,
"cache_creation_input_token_cost_above_272k_tokens": 2.5e-05,
"cache_read_input_token_cost": 1e-06,
"cache_read_input_token_cost_above_272k_tokens": 2e-06,
"output_cost_per_token": 5e-05,
"output_cost_per_token_above_272k_tokens": 7.5e-05,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 1050000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_reasoning": true,
"supports_prompt_caching": true,
"supports_vision": true
},
"bedrock_mantle/openai.gpt-5.5": {
"input_cost_per_token": 5.5e-06,
"input_cost_per_token_above_272k_tokens": 1.1e-05,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -382,6 +382,8 @@ def test_reasoning_with_forced_tool_choice_switches_to_auto():
"us.openai.gpt-5.6-sol",
"global.openai.gpt-5.6-terra",
"bedrock/converse/us.openai.gpt-5.6-luna",
"global.openai.gpt-6-astra",
"bedrock/converse/us.openai.gpt-6-astra",
],
)
def test_reasoning_effort_maps_to_reasoning_effort_for_openai_gpt5_converse(model, local_model_cost_map):
Expand Down Expand Up @@ -412,6 +414,7 @@ def test_reasoning_effort_maps_to_reasoning_effort_for_openai_gpt5_converse(mode
[
"us.openai.gpt-5.6-sol",
"bedrock/converse/global.openai.gpt-5.6-luna",
"us.openai.gpt-6-astra",
],
)
def test_openai_gpt5_converse_never_forwards_thinking(model, local_model_cost_map):
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -98,6 +98,24 @@ class GptProfile(NamedTuple):
),
]

# From the GPT-6 Astra model card: Geo CRIS (us.) and Global CRIS (global.), 272K tiers, cache rates.
GPT_6_PROFILES = [
GptProfile(
model_id="us.openai.gpt-6-astra",
input_cost=1.1e-05, input_cost_above_272k=2.2e-05,
cache_write=1.375e-05, cache_write_above_272k=2.75e-05,
cache_read=1.1e-06, cache_read_above_272k=2.2e-06,
output_cost=5.5e-05, output_cost_above_272k=8.25e-05,
),
GptProfile(
model_id="global.openai.gpt-6-astra",
input_cost=1e-05, input_cost_above_272k=2e-05,
cache_write=1.25e-05, cache_write_above_272k=2.5e-05,
cache_read=1e-06, cache_read_above_272k=2e-06,
output_cost=5e-05, output_cost_above_272k=7.5e-05,
),
]


def _bedrock_response(model, usage):
return ModelResponse(
Expand Down Expand Up @@ -228,3 +246,54 @@ def test_bedrock_gpt_5_6_offers_tools_and_reasoning_effort_but_not_thinking(prof
assert "reasoning_effort" in supported
assert "thinking" not in supported
assert "output_config" not in supported


@pytest.mark.parametrize("profile", GPT_6_PROFILES, ids=lambda p: p.model_id)
def test_bedrock_gpt_6_profiles_route_to_converse(profile, local_model_cost_map):
"""GPT-6 Astra has no in-Region id and no Invoke support on bedrock-runtime; the
us./global. profiles must resolve to Converse, not fall through to Invoke."""
assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "converse"


@pytest.mark.parametrize("profile", GPT_6_PROFILES, ids=lambda p: p.model_id)
def test_bedrock_gpt_6_price_map_matches_model_card(profile, local_model_cost_map):
info = _get_model_info_helper(model=profile.model_id, custom_llm_provider="bedrock")

assert info["mode"] == "chat"
assert info["max_input_tokens"] == 1050000
assert info["max_output_tokens"] == 128000
assert info["input_cost_per_token"] == pytest.approx(profile.input_cost)
assert info["input_cost_per_token_above_272k_tokens"] == pytest.approx(profile.input_cost_above_272k)
assert info["cache_creation_input_token_cost"] == pytest.approx(profile.cache_write)
assert info["cache_creation_input_token_cost_above_272k_tokens"] == pytest.approx(profile.cache_write_above_272k)
assert info["cache_read_input_token_cost"] == pytest.approx(profile.cache_read)
assert info["cache_read_input_token_cost_above_272k_tokens"] == pytest.approx(profile.cache_read_above_272k)
assert info["output_cost_per_token"] == pytest.approx(profile.output_cost)
assert info["output_cost_per_token_above_272k_tokens"] == pytest.approx(profile.output_cost_above_272k)


def test_bedrock_gpt_6_above_272k_tier_applies_to_cost(local_model_cost_map):
response = _bedrock_response(
"bedrock/global.openai.gpt-6-astra",
Usage(prompt_tokens=300000, completion_tokens=1000, total_tokens=301000),
)

cost = completion_cost(
completion_response=response,
model="bedrock/global.openai.gpt-6-astra",
custom_llm_provider="bedrock",
)

assert cost == pytest.approx((300000 * 2e-05) + (1000 * 7.5e-05), rel=1e-9)


@pytest.mark.parametrize("profile", GPT_6_PROFILES, ids=lambda p: p.model_id)
def test_bedrock_gpt_6_offers_tools_and_reasoning_effort_but_not_thinking(profile, local_model_cost_map):
"""Same Converse surface as GPT-5.6: tools and reasoning_effort in, Anthropic thinking out."""
supported = AmazonConverseConfig().get_supported_openai_params(model=f"bedrock/{profile.model_id}")

assert "tools" in supported
assert "tool_choice" in supported
assert "reasoning_effort" in supported
assert "thinking" not in supported
assert "output_config" not in supported
Loading