Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
85 changes: 83 additions & 2 deletions agent/agent_runtime_helpers.py
Original file line number Diff line number Diff line change
Expand Up @@ -2145,6 +2145,32 @@ def plan_cache_sections_for_destination(
return plan.messages, plan.tools


def _is_litellm_route(provider_lower: str, base_url: str) -> bool:
"""True when a route is a LiteLLM proxy, by provider id or host token.

Provider naming varies per install (``litellm``, ``custom:litellm``, or a
bare ``custom`` alias pointed at a LiteLLM host), so both signals are
checked. Both match ``litellm`` as a whole delimited token rather than a
raw substring: ``base_url_hostname``'s own docstring names substring host
matching as the false-positive class to avoid, and a plain
``"litellm" in ...`` grants Anthropic markers to unrelated routes like
``notlitellm.example.com`` or a provider named ``custom:notlitellm``.
A ``litellm`` *path* segment never qualifies — only the host does.
"""
if _has_litellm_token(provider_lower, ":-_/"):
return True
return _has_litellm_token(base_url_hostname(base_url), ".-")


def _has_litellm_token(value: str, delimiters: str) -> bool:
"""True when ``value`` contains ``litellm`` as a whole delimited token."""
if not value:
return False
for delimiter in delimiters:
value = value.replace(delimiter, " ")
return "litellm" in value.split()


def anthropic_prompt_cache_policy(
agent,
*,
Expand Down Expand Up @@ -2269,8 +2295,23 @@ def anthropic_prompt_cache_policy(
# capability declaration instead; explicit false is authoritative too.
# This preserves the runtime model id (and therefore request/cache keys)
# while avoiding unsafe alias-name guesses.
#
# Also consulted for a LiteLLM route on the OpenAI wire: that grant is
# inferred from the provider/host name, so an operator who explicitly
# declares prompt_caching for the route+model must still win over the
# inference — in either direction. Narrowed to the routes the LiteLLM
# branch below can actually grant (chat_completions + Claude): the lookup
# calls get_compatible_custom_providers, which rebuilds its normalized
# view on every call (~1.5ms uncached), and this function runs per
# request destination. Widening it unconditionally regressed the
# non-declaring common case ~200x (7.5us -> 1528us).
custom_prompt_caching = None
if is_anthropic_wire:
_litellm_openai_wire = (
eff_api_mode == "chat_completions"
and is_claude
and _is_litellm_route(provider_lower, eff_base_url)
)
if is_anthropic_wire or _litellm_openai_wire:
try:
from hermes_cli.config import get_custom_provider_model_capability

Expand All @@ -2286,7 +2327,11 @@ def anthropic_prompt_cache_policy(
_cap_exc,
)
if custom_prompt_caching is not None:
return custom_prompt_caching, custom_prompt_caching
# Layout follows the transport, not the declaration: the native
# inner-block form is only honored on the Anthropic Messages wire
# (see the LiteLLM OpenAI-wire branch below for why a top-level
# marker is dropped or 400s on chat_completions).
return custom_prompt_caching, custom_prompt_caching and is_anthropic_wire

# MiniMax-M3 rides MiniMax's server-side automatic prefix cache on the
# Anthropic wire (content-keyed, no marker needed); explicit cache_control
Expand Down Expand Up @@ -2334,6 +2379,42 @@ def anthropic_prompt_cache_policy(
# Third-party Anthropic-compatible gateway.
return True, True

# LiteLLM fronting a Claude model on the OpenAI-compatible wire.
# The branch above only matches LiteLLM in Anthropic proxy mode
# (api_mode == "anthropic_messages"). A LiteLLM deployment that
# exposes /v1/chat/completions instead matched no grant branch above
# and fell through to (False, False): no cache_control is injected, the
# system prompt goes on the wire as a plain string, and the provider
# serves zero cache hits — the entire prompt is re-billed at full price
# every turn. Same failure class already documented above for
# Qwen/DashScope. The endpoint supports Anthropic-style cache_control
# fine; only the provider detection missed it (#84506).
#
# Gated on the Claude family only: a Gemini/GPT/Qwen route through the
# same proxy must not receive markers — some strict OpenAI-wire relays
# reject the cache_control block format outright (cf. the DeepSeek /
# OpenCode exclusion below, #77217).
#
# Envelope layout (native_anthropic=False), matching every other
# OpenAI-wire grant in this function. The native inner-block layout
# writes a TOP-LEVEL msg["cache_control"] on role:tool and
# empty-content messages and relies on the Anthropic adapter to
# relocate it — but that adapter only runs for api_mode ==
# "anthropic_messages" (agent/transports/anthropic.py), and the
# chat_completions transport performs no relocation. On this wire the
# native layout therefore (a) silently loses those breakpoints, spending
# 2 of the 4 available on markers the provider never sees, and (b) when
# LiteLLM relocates a top-level marker itself for an OpenRouter-backed
# Claude route, lands it on an empty text block — the HTTP 400
# "text content blocks must contain" shape handled in
# agent/anthropic_adapter.py (#69512).
#
# Gated on chat_completions explicitly rather than `not
# is_anthropic_wire`: codex_responses / bedrock_converse are separate
# transports with their own marker handling and must not be swept in.
if _litellm_openai_wire:
return True, False

# MiniMax on its Anthropic-compatible endpoint serves its own
# model family (MiniMax-M2.7, M2.5, M2.1, M2) with documented
# cache_control support (0.1× read pricing, 5-minute TTL). The
Expand Down
272 changes: 272 additions & 0 deletions tests/run_agent/test_anthropic_prompt_cache_policy.py
Original file line number Diff line number Diff line change
Expand Up @@ -550,6 +550,278 @@ def test_deepseek_on_openrouter_does_not_cache(self):
assert agent._anthropic_prompt_cache_policy() == (False, False)


class TestLiteLLMOpenAIWire:
"""LiteLLM fronting a Claude model on the OpenAI-compatible wire (#84506).

A LiteLLM proxy exposing /v1/chat/completions (api_mode ==
"chat_completions", /v1/messages returns 404) previously matched no
grant branch and fell through to (False, False): zero cache hits, the
full prompt re-billed every turn. The endpoint accepts Anthropic-style
cache_control fine — only the provider detection missed it. Claude gets
the grant with the envelope layout (the only layout honored on this
wire); non-Claude models routed through the same proxy get nothing
(they may not tolerate the marker block format).
"""

@pytest.mark.parametrize(
"provider,base_url",
[
# Provider-string signal: names vary per install.
("litellm", "https://my-litellm-host.example.com/v1"),
("custom:litellm", "https://my-litellm-host.example.com/v1"),
# Host signal: bare `custom` alias pointed at a LiteLLM host.
("custom", "https://litellm.internal.example.com/v1"),
# Host signal, hyphen-delimited label (self-hosted naming).
("custom", "https://my-litellm-gw.internal.example.com/v1"),
],
)
@pytest.mark.parametrize(
"model",
[
"claude-opus-4.8",
"anthropic/claude-sonnet-4.6",
],
)
def test_claude_on_litellm_openai_wire_caches_with_envelope_layout(
self, provider, base_url, model
):
agent = _make_agent(
provider=provider,
base_url=base_url,
api_mode="chat_completions",
model=model,
)
assert agent._anthropic_prompt_cache_policy() == (True, False)

@pytest.mark.parametrize(
"model",
[
"openai/gpt-5.4",
"gemini-2.5-pro",
"qwen3.6-plus",
"deepseek-v4-pro",
],
)
def test_non_claude_on_litellm_openai_wire_does_not_cache(self, model):
# No over-reach: a Gemini/GPT/Qwen/DeepSeek route through the same
# LiteLLM proxy must not receive Anthropic cache_control markers.
agent = _make_agent(
provider="litellm",
base_url="https://litellm.internal.example.com/v1",
api_mode="chat_completions",
model=model,
)
assert agent._anthropic_prompt_cache_policy() == (False, False)

def test_litellm_claude_operator_disable_still_wins(self):
# prompt_caching.cache_ttl: false — the _cache_disabled early return
# must survive the new branch.
agent = _make_agent(
provider="litellm",
base_url="https://litellm.internal.example.com/v1",
api_mode="chat_completions",
model="claude-opus-4.8",
)
agent._cache_disabled = True
assert agent._anthropic_prompt_cache_policy() == (False, False)

def test_litellm_in_anthropic_proxy_mode_still_uses_native_layout(self):
# Adjacent behavior: LiteLLM reached over the native Anthropic wire
# keeps hitting the pre-existing is_anthropic_wire branch (True, True).
agent = _make_agent(
provider="litellm",
base_url="https://litellm.internal.example.com",
api_mode="anthropic_messages",
model="claude-opus-4.8",
)
assert agent._anthropic_prompt_cache_policy() == (True, True)

@pytest.mark.parametrize(
"base_url",
[
# "litellm" as a substring of a longer label is NOT a LiteLLM host.
"https://notlitellm.attacker.example/v1",
"https://foolitellmbar.example/v1",
# A "litellm" PATH segment on an unrelated host must not qualify.
"https://gateway.attacker.example/litellm/v1",
],
)
def test_litellm_lookalike_hosts_do_not_cache(self, base_url):
# Host matching is label-token-wise, not substring: a Claude-named
# model on an unrelated strict OpenAI-wire relay must not receive
# Anthropic markers (it may reject the block format, cf. #77217).
agent = _make_agent(
provider="custom",
base_url=base_url,
api_mode="chat_completions",
model="claude-opus-4.8",
)
assert agent._anthropic_prompt_cache_policy() == (False, False)

@pytest.mark.parametrize(
"provider", ["custom:notlitellm", "notlitellm", "mylitellmthing"]
)
def test_litellm_lookalike_provider_names_do_not_cache(self, provider):
# The provider signal is token-wise for the same reason as the host:
# a user-named provider that merely contains "litellm" is not a
# LiteLLM route and must not be handed Anthropic markers.
agent = _make_agent(
provider=provider,
base_url="https://gateway.attacker.example/v1",
api_mode="chat_completions",
model="claude-opus-4.8",
)
assert agent._anthropic_prompt_cache_policy() == (False, False)

@pytest.mark.parametrize(
"provider", ["litellm", "custom:litellm", "litellm-router", "LiteLLM"]
)
def test_litellm_provider_spellings_still_cache(self, provider):
# ...while every real spelling of a LiteLLM provider id still matches.
agent = _make_agent(
provider=provider,
base_url="https://gateway.internal.example/v1",
api_mode="chat_completions",
model="claude-opus-4.8",
)
assert agent._anthropic_prompt_cache_policy() == (True, False)

@pytest.mark.parametrize(
"api_mode", ["codex_responses", "bedrock_converse", "codex_app_server"]
)
def test_litellm_claude_on_other_transports_does_not_cache(self, api_mode):
# The grant is scoped to chat_completions. Other transports carry
# their own marker handling and must not be swept in by a blanket
# "not anthropic_messages" gate.
agent = _make_agent(
provider="custom:litellm",
base_url="https://litellm.internal.example.com/v1",
api_mode=api_mode,
model="claude-opus-4.8",
)
assert agent._anthropic_prompt_cache_policy() == (False, False)

def test_operator_capability_declaration_overrides_litellm_inference(self):
# The LiteLLM grant is inferred from the provider/host name, so an
# explicit per-model declaration must still win — otherwise an
# operator who turned caching off for a known-broken route on this
# proxy is silently overridden.
agent = _make_agent(
provider="custom:litellm",
base_url="https://litellm.internal.example.com/v1",
api_mode="chat_completions",
model="claude-opus-4.8",
)
agent._custom_providers = [
{
"name": "litellm",
"base_url": "https://litellm.internal.example.com/v1",
"models": {"claude-opus-4.8": {"prompt_caching": False}},
}
]
assert agent._anthropic_prompt_cache_policy() == (False, False)

def test_capability_declared_true_keeps_envelope_layout_on_openai_wire(self):
# An explicit prompt_caching: true must not promote the request to the
# native inner-block layout on chat_completions — the layout follows
# the transport, and a top-level marker is dropped there.
agent = _make_agent(
provider="custom:litellm",
base_url="https://litellm.internal.example.com/v1",
api_mode="chat_completions",
model="claude-opus-4.8",
)
agent._custom_providers = [
{
"name": "litellm",
"base_url": "https://litellm.internal.example.com/v1",
"models": {"claude-opus-4.8": {"prompt_caching": True}},
}
]
assert agent._anthropic_prompt_cache_policy() == (True, False)

def test_capability_declared_false_wins_over_openrouter_grant(self):
# A litellm-named provider pointed at OpenRouter previously took the
# OpenRouter branch and ignored an explicit per-model opt-out, because
# the capability lookup was gated on the Anthropic wire. The operator's
# declaration now wins on this wire too.
agent = _make_agent(
provider="custom:litellm",
base_url="https://openrouter.ai/api/v1",
api_mode="chat_completions",
model="claude-opus-4.8",
)
agent._custom_providers = [
{
"name": "litellm",
"base_url": "https://openrouter.ai/api/v1",
"models": {"claude-opus-4.8": {"prompt_caching": False}},
}
]
assert agent._anthropic_prompt_cache_policy() == (False, False)

def test_litellm_provider_on_lookalike_host_still_grants(self):
# Precedence is intentional and pinned: the provider id is an
# independent signal, so an explicitly litellm-named provider grants
# even when the HOST is a lookalike. Only the host-derived signal is
# token-gated (see test_litellm_lookalike_hosts_do_not_cache, which
# uses provider="custom").
agent = _make_agent(
provider="custom:litellm",
base_url="https://notlitellm.attacker.example/v1",
api_mode="chat_completions",
model="claude-opus-4.8",
)
assert agent._anthropic_prompt_cache_policy() == (True, False)

def test_litellm_openai_wire_emits_no_top_level_marker(self):
# Wire-shape contract, not just the policy tuple: on chat_completions
# every breakpoint must land INSIDE a content part. A top-level
# msg["cache_control"] is never relocated on this transport, so it is
# both a lost breakpoint and (once a relay relocates it onto an empty
# assistant turn) the HTTP 400 empty-text-block shape (#69512).
from agent.agent_runtime_helpers import plan_cache_sections_for_destination

messages = [
{"role": "system", "content": "SYSTEM " * 200},
{"role": "user", "content": "go"},
{
"role": "assistant",
"content": "",
"tool_calls": [
{
"id": "c0",
"type": "function",
"function": {"name": "terminal", "arguments": "{}"},
}
],
},
{"role": "tool", "tool_call_id": "c0", "content": "output " * 100},
{"role": "assistant", "content": "done"},
]
planned, _tools = plan_cache_sections_for_destination(
messages,
None,
provider="custom:litellm",
base_url="https://litellm.internal.example.com/v1",
api_mode="chat_completions",
model="claude-opus-4.8",
cache_disabled=False,
cache_ttl="5m",
)
assert not [m for m in planned if "cache_control" in m], (
"no breakpoint may sit on the message envelope on the OpenAI wire"
)
inner = [
m
for m in planned
if isinstance(m.get("content"), list)
for part in m["content"]
if isinstance(part, dict) and "cache_control" in part
]
assert inner, "the OpenAI-wire grant must still place real breakpoints"


class TestNousPortalAnthropicWire:
def test_portal_claude_on_the_messages_wire_uses_the_native_layout(self):
agent = _make_agent(
Expand Down
Loading