Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 23 additions & 5 deletions litellm/llms/bedrock/chat/converse_transformation.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@

import copy
import json
import re
import time
import types
from collections.abc import Mapping
Expand Down Expand Up @@ -104,6 +105,7 @@
"bash_",
"text_editor_",
]
BEDROCK_OPENAI_COMPAT_MIN_MAX_TOKENS: Final = 16

# Beta header patterns that are not supported by Bedrock Converse API
# These will be filtered out to prevent errors
Expand Down Expand Up @@ -292,6 +294,14 @@ def _validate_request_metadata(self, metadata: dict) -> None:
llm_provider="bedrock",
)

@staticmethod
def _requires_min_max_tokens(model: str) -> bool:
return re.search(r"openai\.gpt-\d|xai\.grok-", model) is not None

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P1 Opaque profiles bypass clamp

Hardcoded matching violates the directive to store model capabilities in metadata and misses opaque profiles, so GPT/Grok requests below 16 are rejected upstream

Rule Used: What: Do not hardcode model-specific flags in the codebase. Instead, put them in model_prices_and_context_window.json and then read them in via get_model_info Why: Prevents need for users to upgrade litellm each time a new model supports this featu... (source)

Knowledge Base Used: Provider adapters and capabilities

Note: If this suggestion doesn't match your team's coding style, reply to this and let me know. I'll remember it for next time!

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Backport of #41870, where this was withdrawn. The regex clamps ARNs embedding the model id; opaque application profiles carry none, so get_model_info cannot resolve them

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

You're right. In this backport, matching the model string is intentional: it covers inference-profile ARNs that embed openai.gpt-* or xai.grok-*, while opaque application-inference-profile ARNs contain no resolvable model identifier for get_model_info to use. Given that limitation and the behavior established by #41870, this comment does not apply. No change needed.


@staticmethod
def _is_openai_gpt_reasoning_model(model: str) -> bool:
return re.search(r"openai\.gpt-\d", model) is not None

Comment on lines +303 to +304

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Hardcoded reasoning capability

_is_openai_gpt_reasoning_model decides reasoning support from the openai.gpt-<digit> model name. That result controls both the reasoning.effort mapping and whether thinking is discarded. This violates the repository directive that model-specific capabilities must be stored in model_prices_and_context_window.json and queried through model metadata. The repository requirement must be satisfied before merging.

Rule Used: What: Do not hardcode model-specific flags in the codebase. Instead, put them in model_prices_and_context_window.json and then read them in via get_model_info Why: Prevents need for users to upgrade litellm each time a new model supports this featu... (source)

Note: If this suggestion doesn't match your team's coding style, reply to this and let me know. I'll remember it for next time!

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This mirrors main's merged #31884. A stable backport should not diverge from main, and the family pattern covers new GPT releases without a cost-map entry

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

That’s fair. For this PR, matching main’s merged #31884 behavior is the correct backport constraint, and the family-level match is intentionally forward-compatible with future openai.gpt-<digit> Bedrock models without requiring a cost-map update for each release. I’m withdrawing this as a merge blocker; the metadata-vs-family-gate convention can be revisited separately if main changes.

def _is_nova_2_model(self, model: str) -> bool:
"""
Check if the model is a Nova 2 model that supports reasoningConfig.
Expand Down Expand Up @@ -422,14 +432,14 @@ def _handle_reasoning_effort_parameter(self, model: str, reasoning_effort: str,
Handle the reasoning_effort parameter based on the model type.

- GPT-OSS models: passed through unchanged via additionalModelRequestFields.
- OpenAI GPT-5.x models: mapped to ``reasoning.effort`` via additionalModelRequestFields.
- OpenAI GPT-5.x and GPT-6 models: mapped to ``reasoning.effort`` via additionalModelRequestFields.
- Nova 2 models: transformed to reasoningConfig.
- Anthropic models: mapped to ``thinking`` (and ``output_config.effort`` on
adaptive Claude 4.6 / 4.7).
"""
if "gpt-oss" in model:
optional_params["reasoning_effort"] = reasoning_effort
elif "openai.gpt-5" in model:
elif self._is_openai_gpt_reasoning_model(model):
reasoning: Final[BedrockConverseGptReasoningEffortBlock] = {"effort": reasoning_effort}
optional_params["reasoning"] = reasoning
elif self._is_nova_2_model(model):
Expand Down Expand Up @@ -563,7 +573,11 @@ def get_supported_openai_params(self, model: str) -> list[str]:
# only anthropic and mistral support tool choice config. otherwise (E.g. cohere) will fail the call - https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_ToolChoice.html
supported_params.append("tool_choice")

if "gpt-oss" in model or "openai.gpt-5" in model or "openai.gpt-5" in base_model:
if (
"gpt-oss" in model
or self._is_openai_gpt_reasoning_model(model)
or self._is_openai_gpt_reasoning_model(base_model)
):
supported_params.append("reasoning_effort")
elif self._is_nova_2_model(model):
# Nova 2 models support reasoning_effort (transformed to reasoningConfig)
Expand Down Expand Up @@ -874,7 +888,11 @@ def map_openai_params(
is_thinking_enabled=is_thinking_enabled,
)
if param == "max_tokens" or param == "max_completion_tokens":
optional_params["maxTokens"] = value
optional_params["maxTokens"] = (
max(value, BEDROCK_OPENAI_COMPAT_MIN_MAX_TOKENS)
if isinstance(value, int) and self._requires_min_max_tokens(model)
else value
)
if param == "stream":
optional_params["stream"] = value
if param == "stop":
Expand Down Expand Up @@ -911,7 +929,7 @@ def map_openai_params(
optional_params["_parallel_tool_use_config"] = {
"tool_choice": {"type": "auto", "disable_parallel_tool_use": not value}
}
if param == "thinking" and "openai.gpt-5" not in model:
if param == "thinking" and not self._is_openai_gpt_reasoning_model(model):
if (
isinstance(value, dict)
and value.get("type") == "adaptive"
Expand Down
4 changes: 2 additions & 2 deletions pyproject.toml
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
[project]
name = "litellm"
version = "1.100.1"
version = "1.100.2"
description = "Library to easily interface with LLM API providers"
readme = "README.md"
requires-python = ">=3.10, <3.15"
Expand Down Expand Up @@ -311,7 +311,7 @@ members = ["enterprise", "litellm-proxy-extras"]
profile = "black"

[tool.commitizen]
version = "1.100.1"
version = "1.100.2"
version_files = [
"pyproject.toml:^version",
]
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -375,12 +375,42 @@ def test_reasoning_with_forced_tool_choice_switches_to_auto():
assert optional_params["tool_choice"] == {"auto": {}}


@pytest.mark.parametrize(
"model, param, value, expected_max_tokens",
[
("us.openai.gpt-6-astra", "max_tokens", 1, 16),
("us.openai.gpt-6-astra", "max_completion_tokens", 1, 16),
("us.openai.gpt-6-astra", "max_tokens", 64, 64),
("us.xai.grok-4.6", "max_tokens", 1, 16),
("global.xai.grok-4.6", "max_completion_tokens", 1, 16),
("us.xai.grok-4.6", "max_tokens", 32, 32),
("anthropic.claude-sonnet-4-5-20250929-v1:0", "max_tokens", 1, 1),
("arn:aws:bedrock:us-east-1:123456789012:inference-profile/us.openai.gpt-6-astra", "max_tokens", 1, 16),
("arn:aws:bedrock:us-east-1:123456789012:inference-profile/global.xai.grok-4.6", "max_tokens", 1, 16),
("arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/abc123xyz", "max_tokens", 1, 1),
],
)
def test_map_openai_params_enforces_minimum_max_tokens_for_openai_compat_models(
model: str, param: str, value: int, expected_max_tokens: int
):
optional_params = AmazonConverseConfig().map_openai_params(
non_default_params={param: value},
optional_params={},
model=model,
drop_params=False,
)

assert optional_params["maxTokens"] == expected_max_tokens


@pytest.mark.parametrize(
"model",
[
"us.openai.gpt-5.6-sol",
"global.openai.gpt-5.6-terra",
"bedrock/converse/us.openai.gpt-5.6-luna",
"us.openai.gpt-6-astra",
"bedrock/converse/global.openai.gpt-6-astra",
],
)
def test_reasoning_effort_maps_to_reasoning_effort_for_openai_gpt5_converse(model, local_model_cost_map):
Expand Down Expand Up @@ -411,6 +441,7 @@ def test_reasoning_effort_maps_to_reasoning_effort_for_openai_gpt5_converse(mode
[
"us.openai.gpt-5.6-sol",
"bedrock/converse/global.openai.gpt-5.6-luna",
"us.openai.gpt-6-astra",
],
)
def test_openai_gpt5_converse_never_forwards_thinking(model, local_model_cost_map):
Expand Down
2 changes: 1 addition & 1 deletion uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

Loading