diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index fa24f8be8932..e5e515b580fb 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -115,6 +115,12 @@ ] +def _is_openai_frontier_model(model: str) -> bool: + """OpenAI GPT-5.x / GPT-6 on Bedrock (``openai.gpt-5.6-sol``, ``openai.gpt-6-astra``): reasoning goes + through ``reasoning.effort``. ``openai.gpt-oss-*`` is excluded; it takes ``reasoning_effort`` as-is.""" + return "openai.gpt-" in model and "gpt-oss" not in model + + class AmazonConverseConfig(BaseConfig): """ Reference - https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_Converse.html @@ -430,7 +436,7 @@ def _handle_reasoning_effort_parameter(self, model: str, reasoning_effort: str, """ if "gpt-oss" in model: optional_params["reasoning_effort"] = reasoning_effort - elif "openai.gpt-5" in model: + elif _is_openai_frontier_model(model): reasoning: Final[BedrockConverseGptReasoningEffortBlock] = {"effort": reasoning_effort} optional_params["reasoning"] = reasoning elif self._is_nova_2_model(model): @@ -564,7 +570,7 @@ def get_supported_openai_params(self, model: str) -> list[str]: # only anthropic and mistral support tool choice config. otherwise (E.g. cohere) will fail the call - https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_ToolChoice.html supported_params.append("tool_choice") - if "gpt-oss" in model or "openai.gpt-5" in model or "openai.gpt-5" in base_model: + if "gpt-oss" in model or _is_openai_frontier_model(model) or _is_openai_frontier_model(base_model): supported_params.append("reasoning_effort") elif self._is_nova_2_model(model): # Nova 2 models support reasoning_effort (transformed to reasoningConfig) @@ -920,7 +926,7 @@ def map_openai_params( optional_params["_parallel_tool_use_config"] = { "tool_choice": {"type": "auto", "disable_parallel_tool_use": not value} } - if param == "thinking" and "openai.gpt-5" not in model: + if param == "thinking" and not _is_openai_frontier_model(model): if ( isinstance(value, dict) and value.get("type") == "adaptive" diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 54ebdc85be9e..0f13bc8cc195 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -55196,6 +55196,60 @@ "supports_reasoning": true, "supports_vision": true }, + "us.openai.gpt-6-astra": { + "input_cost_per_token": 1.1e-05, + "input_cost_per_token_above_272k_tokens": 2.2e-05, + "cache_creation_input_token_cost": 1.375e-05, + "cache_creation_input_token_cost_above_272k_tokens": 2.75e-05, + "cache_read_input_token_cost": 1.1e-06, + "cache_read_input_token_cost_above_272k_tokens": 2.2e-06, + "output_cost_per_token": 5.5e-05, + "output_cost_per_token_above_272k_tokens": 8.25e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_prompt_caching": true, + "supports_vision": true + }, + "global.openai.gpt-6-astra": { + "input_cost_per_token": 1e-05, + "input_cost_per_token_above_272k_tokens": 2e-05, + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_272k_tokens": 2.5e-05, + "cache_read_input_token_cost": 1e-06, + "cache_read_input_token_cost_above_272k_tokens": 2e-06, + "output_cost_per_token": 5e-05, + "output_cost_per_token_above_272k_tokens": 7.5e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_prompt_caching": true, + "supports_vision": true + }, "bedrock_mantle/openai.gpt-5.5": { "input_cost_per_token": 5.5e-06, "input_cost_per_token_above_272k_tokens": 1.1e-05, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 54ebdc85be9e..0f13bc8cc195 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -55196,6 +55196,60 @@ "supports_reasoning": true, "supports_vision": true }, + "us.openai.gpt-6-astra": { + "input_cost_per_token": 1.1e-05, + "input_cost_per_token_above_272k_tokens": 2.2e-05, + "cache_creation_input_token_cost": 1.375e-05, + "cache_creation_input_token_cost_above_272k_tokens": 2.75e-05, + "cache_read_input_token_cost": 1.1e-06, + "cache_read_input_token_cost_above_272k_tokens": 2.2e-06, + "output_cost_per_token": 5.5e-05, + "output_cost_per_token_above_272k_tokens": 8.25e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_prompt_caching": true, + "supports_vision": true + }, + "global.openai.gpt-6-astra": { + "input_cost_per_token": 1e-05, + "input_cost_per_token_above_272k_tokens": 2e-05, + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_272k_tokens": 2.5e-05, + "cache_read_input_token_cost": 1e-06, + "cache_read_input_token_cost_above_272k_tokens": 2e-06, + "output_cost_per_token": 5e-05, + "output_cost_per_token_above_272k_tokens": 7.5e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_prompt_caching": true, + "supports_vision": true + }, "bedrock_mantle/openai.gpt-5.5": { "input_cost_per_token": 5.5e-06, "input_cost_per_token_above_272k_tokens": 1.1e-05, diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index f0e361ceb885..2b58046aac6f 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -382,6 +382,8 @@ def test_reasoning_with_forced_tool_choice_switches_to_auto(): "us.openai.gpt-5.6-sol", "global.openai.gpt-5.6-terra", "bedrock/converse/us.openai.gpt-5.6-luna", + "global.openai.gpt-6-astra", + "bedrock/converse/us.openai.gpt-6-astra", ], ) def test_reasoning_effort_maps_to_reasoning_effort_for_openai_gpt5_converse(model, local_model_cost_map): @@ -412,6 +414,7 @@ def test_reasoning_effort_maps_to_reasoning_effort_for_openai_gpt5_converse(mode [ "us.openai.gpt-5.6-sol", "bedrock/converse/global.openai.gpt-5.6-luna", + "us.openai.gpt-6-astra", ], ) def test_openai_gpt5_converse_never_forwards_thinking(model, local_model_cost_map): diff --git a/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py b/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py index fda3c8ceb8f3..a396b71fd52c 100644 --- a/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py +++ b/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py @@ -98,6 +98,24 @@ class GptProfile(NamedTuple): ), ] +# From the GPT-6 Astra model card: Geo CRIS (us.) and Global CRIS (global.), 272K tiers, cache rates. +GPT_6_PROFILES = [ + GptProfile( + model_id="us.openai.gpt-6-astra", + input_cost=1.1e-05, input_cost_above_272k=2.2e-05, + cache_write=1.375e-05, cache_write_above_272k=2.75e-05, + cache_read=1.1e-06, cache_read_above_272k=2.2e-06, + output_cost=5.5e-05, output_cost_above_272k=8.25e-05, + ), + GptProfile( + model_id="global.openai.gpt-6-astra", + input_cost=1e-05, input_cost_above_272k=2e-05, + cache_write=1.25e-05, cache_write_above_272k=2.5e-05, + cache_read=1e-06, cache_read_above_272k=2e-06, + output_cost=5e-05, output_cost_above_272k=7.5e-05, + ), +] + def _bedrock_response(model, usage): return ModelResponse( @@ -228,3 +246,54 @@ def test_bedrock_gpt_5_6_offers_tools_and_reasoning_effort_but_not_thinking(prof assert "reasoning_effort" in supported assert "thinking" not in supported assert "output_config" not in supported + + +@pytest.mark.parametrize("profile", GPT_6_PROFILES, ids=lambda p: p.model_id) +def test_bedrock_gpt_6_profiles_route_to_converse(profile, local_model_cost_map): + """GPT-6 Astra has no in-Region id and no Invoke support on bedrock-runtime; the + us./global. profiles must resolve to Converse, not fall through to Invoke.""" + assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "converse" + + +@pytest.mark.parametrize("profile", GPT_6_PROFILES, ids=lambda p: p.model_id) +def test_bedrock_gpt_6_price_map_matches_model_card(profile, local_model_cost_map): + info = _get_model_info_helper(model=profile.model_id, custom_llm_provider="bedrock") + + assert info["mode"] == "chat" + assert info["max_input_tokens"] == 1050000 + assert info["max_output_tokens"] == 128000 + assert info["input_cost_per_token"] == pytest.approx(profile.input_cost) + assert info["input_cost_per_token_above_272k_tokens"] == pytest.approx(profile.input_cost_above_272k) + assert info["cache_creation_input_token_cost"] == pytest.approx(profile.cache_write) + assert info["cache_creation_input_token_cost_above_272k_tokens"] == pytest.approx(profile.cache_write_above_272k) + assert info["cache_read_input_token_cost"] == pytest.approx(profile.cache_read) + assert info["cache_read_input_token_cost_above_272k_tokens"] == pytest.approx(profile.cache_read_above_272k) + assert info["output_cost_per_token"] == pytest.approx(profile.output_cost) + assert info["output_cost_per_token_above_272k_tokens"] == pytest.approx(profile.output_cost_above_272k) + + +def test_bedrock_gpt_6_above_272k_tier_applies_to_cost(local_model_cost_map): + response = _bedrock_response( + "bedrock/global.openai.gpt-6-astra", + Usage(prompt_tokens=300000, completion_tokens=1000, total_tokens=301000), + ) + + cost = completion_cost( + completion_response=response, + model="bedrock/global.openai.gpt-6-astra", + custom_llm_provider="bedrock", + ) + + assert cost == pytest.approx((300000 * 2e-05) + (1000 * 7.5e-05), rel=1e-9) + + +@pytest.mark.parametrize("profile", GPT_6_PROFILES, ids=lambda p: p.model_id) +def test_bedrock_gpt_6_offers_tools_and_reasoning_effort_but_not_thinking(profile, local_model_cost_map): + """Same Converse surface as GPT-5.6: tools and reasoning_effort in, Anthropic thinking out.""" + supported = AmazonConverseConfig().get_supported_openai_params(model=f"bedrock/{profile.model_id}") + + assert "tools" in supported + assert "tool_choice" in supported + assert "reasoning_effort" in supported + assert "thinking" not in supported + assert "output_config" not in supported