Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 7 additions & 3 deletions litellm/llms/bedrock/common_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,7 @@


_ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs"
_OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.")

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Hardcoded Model Capability

This regex bypasses model metadata, violating the requirement that model-specific flags use model_prices_and_context_window.json. This requirement must be satisfied before merging

Rule Used: What: Do not hardcode model-specific flags in the codebase. Instead, put them in model_prices_and_context_window.json and then read them in via get_model_info Why: Prevents need for users to upgrade litellm each time a new model supports this featu... (source)

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Converse rejects cachePoint for every Bedrock OpenAI-family model: a provider API contract, not a per-model capability. The map's supports_prompt_caching stays true for implicit cache-read pricing

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

You're right — I conflated two separate capabilities. supports_prompt_caching in the cost map controls cached-token accounting/pricing, including implicit cache reads; it is not a declaration that the Converse wire format accepts an explicit cachePoint. AWS documents OpenAI Bedrock caching as a separate model-family/API contract, so encoding this gate in pricing metadata would incorrectly couple protocol serialization to billing data. The early return in litellm/llms/bedrock/common_utils.py is therefore appropriate, and I withdraw the hardcoded-capability finding.



def error_response_text(response: httpx.Response) -> str:
Expand Down Expand Up @@ -878,16 +879,19 @@ def bedrock_model_accepts_cache_points(model: str | None) -> bool:
"""
Whether Converse ``cachePoint`` blocks may be sent to this model.

Bedrock rejects requests carrying cachePoint blocks for models without prompt
caching support ("You invoked an unsupported model or your request did not allow
prompt caching"), so a model whose cost-map entry does not declare
OpenAI-family models only support implicit caching and never accept explicit
``cachePoint`` blocks. Bedrock rejects requests carrying cachePoint blocks for
models without prompt caching support ("You invoked an unsupported model or your
request did not allow prompt caching"), so a model whose cost-map entry does not declare
``supports_prompt_caching`` must not receive them. A model absent from the map
(an application inference profile ARN, a model newer than the map) keeps emitting
so existing caching setups never silently degrade. ``litellm.utils.supports_prompt_caching``
is not reusable here: it returns False for unmapped models, the opposite polarity.
"""
if model is None:
return True
if _OPENAI_FAMILY_MODEL_RE.search(model):
return False
entries: Final = tuple(
entry
for candidate in (model, get_bedrock_base_model(model))
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -1189,17 +1189,24 @@ def test_get_supported_openai_params_bedrock_converse():


@pytest.mark.parametrize(
"tools, expected_marker",
"tools, model, expected_marker",
[
pytest.param(
[{"type": "function", "function": {"name": "f", "parameters": {"type": "object", "properties": {}}}}],
"anthropic.claude-sonnet-4-5-20250929-v1:0",
"dep-bedrock",
id="tools-present-so-the-cachepoint-is-placed",
),
pytest.param(None, None, id="no-tools-so-nothing-is-placed"),
pytest.param(None, "anthropic.claude-sonnet-4-5-20250929-v1:0", None, id="no-tools-so-nothing-is-placed"),
pytest.param(
[{"type": "function", "function": {"name": "f", "parameters": {"type": "object", "properties": {}}}}],
"global.openai.gpt-6-astra",
None,
id="openai-family-implicit-caching-only",
),
],
)
def test_tool_config_cachepoint_is_credited_only_where_it_is_placed(tools, expected_marker):
def test_tool_config_cachepoint_is_credited_only_where_it_is_placed(tools, model, expected_marker):
"""Spend attribution credits the gateway for breakpoints it placed, and a tool_config
point becomes one here or nowhere.

Expand All @@ -1213,7 +1220,7 @@ def test_tool_config_cachepoint_is_credited_only_where_it_is_placed(tools, expec
optional_params["tools"] = tools

data = AmazonConverseConfig()._transform_request_helper(
model="anthropic.claude-sonnet-4-5-20250929-v1:0",
model=model,
system_content_blocks=[],
optional_params=optional_params,
messages=[{"role": "user", "content": "hi"}],
Expand Down Expand Up @@ -5591,6 +5598,9 @@ def test_cache_control_injection_tool_config_drops_ttl_for_unsupported_model():
True,
id="unmapped-arn-keeps-emitting",
),
pytest.param("global.openai.gpt-6-astra", False, id="openai-family-implicit-caching-only"),
pytest.param("openai.gpt-oss-120b-1:0", False, id="openai-gpt-oss"),
pytest.param("us.openai.gpt-99-unmapped", False, id="unmapped-openai-family-still-suppressed"),
],
)
def test_cache_points_emitted_only_for_models_that_support_prompt_caching(model, expects_cache_points, monkeypatch):
Expand Down
Loading