Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 11 additions & 1 deletion litellm/llms/bedrock/chat/converse_transformation.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@

import copy
import json
import re
import time
import types
from collections.abc import Mapping
Expand Down Expand Up @@ -105,6 +106,7 @@
"bash_",
"text_editor_",
]
BEDROCK_OPENAI_COMPAT_MIN_MAX_TOKENS: Final = 16

# Beta header patterns that are not supported by Bedrock Converse API
# These will be filtered out to prevent errors
Expand Down Expand Up @@ -293,6 +295,10 @@ def _validate_request_metadata(self, metadata: dict) -> None:
llm_provider="bedrock",
)

@staticmethod
def _requires_min_max_tokens(model: str) -> bool:
return re.search(r"openai\.gpt-\d|xai\.grok-", model) is not None
Comment on lines +298 to +300

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Hardcoded model capability detection

This identifies model families with hardcoded name patterns. Repository rules require model-specific flags to live in model_prices_and_context_window.json and be read through get_model_info, so this must be addressed before merging

Rule Used: What: Do not hardcode model-specific flags in the codebase. Instead, put them in model_prices_and_context_window.json and then read them in via get_model_info Why: Prevents need for users to upgrade litellm each time a new model supports this featu... (source)

Knowledge Base Used: Provider adapters and capabilities


def _is_nova_2_model(self, model: str) -> bool:
"""
Check if the model is a Nova 2 model that supports reasoningConfig.
Expand Down Expand Up @@ -883,7 +889,11 @@ def map_openai_params(
is_thinking_enabled=is_thinking_enabled,
)
if param == "max_tokens" or param == "max_completion_tokens":
optional_params["maxTokens"] = value
optional_params["maxTokens"] = (
max(value, BEDROCK_OPENAI_COMPAT_MIN_MAX_TOKENS)

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Low: Token rate-limit under-reservation

The rate-limit pre-call hook reserves the caller's original max_tokens value, but this later raises it to 16. An authenticated caller can submit concurrent requests with max_tokens: 1 and receive up to 16 output tokens per one token reserved, exceeding TPM or OTPM limits before post-call reconciliation. Normalize the effective cap before rate-limit hooks run, or make the limiter apply the same model-family minimum.

if isinstance(value, int) and self._requires_min_max_tokens(model)
else value
)
if param == "stream":
optional_params["stream"] = value
if param == "stop":
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -375,6 +375,34 @@ def test_reasoning_with_forced_tool_choice_switches_to_auto():
assert optional_params["tool_choice"] == {"auto": {}}


@pytest.mark.parametrize(
"model, param, value, expected_max_tokens",
[
("us.openai.gpt-6-astra", "max_tokens", 1, 16),
("us.openai.gpt-6-astra", "max_completion_tokens", 1, 16),
("us.openai.gpt-6-astra", "max_tokens", 64, 64),
("us.xai.grok-4.6", "max_tokens", 1, 16),
("global.xai.grok-4.6", "max_completion_tokens", 1, 16),
("us.xai.grok-4.6", "max_tokens", 32, 32),
("anthropic.claude-sonnet-4-5-20250929-v1:0", "max_tokens", 1, 1),
("arn:aws:bedrock:us-east-1:123456789012:inference-profile/us.openai.gpt-6-astra", "max_tokens", 1, 16),
("arn:aws:bedrock:us-east-1:123456789012:inference-profile/global.xai.grok-4.6", "max_tokens", 1, 16),
("arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/abc123xyz", "max_tokens", 1, 1),
],
)
def test_map_openai_params_enforces_minimum_max_tokens_for_openai_compat_models(
model: str, param: str, value: int, expected_max_tokens: int
):
optional_params = AmazonConverseConfig().map_openai_params(
non_default_params={param: value},
optional_params={},
model=model,
drop_params=False,
)

assert optional_params["maxTokens"] == expected_max_tokens


@pytest.mark.parametrize(
"model",
[
Expand Down
14 changes: 7 additions & 7 deletions uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

Loading