Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
31 changes: 27 additions & 4 deletions agent/auxiliary_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -1703,6 +1703,11 @@ def __init__(self, real_client: Any, model: str, is_oauth: bool = False, base_ur
self._client = real_client
self._model = model
self._is_oauth = is_oauth
# AnthropicBedrock detection by class name: the anthropic SDK is an optional extra, and
# this module must stay importable without it. Only Bedrock's InvokeModel endpoint rejects
# the translated output_config.format (#124923); direct/partner Anthropic and Messages-wire
# gateways accept it.
self._is_bedrock_client = type(real_client).__name__ == "AnthropicBedrock"
# Caller URL first; fall back to the SDK client's host only for Nous Portal — a blanket
# fallback would flip MiniMax/Zhipu aux adapters to third-party handling (strips thinking sigs).
self._base_url = base_url or None
Expand Down Expand Up @@ -1770,12 +1775,30 @@ def create(self, **kwargs) -> Any:
# The adapter builds the Messages body from a fixed allow-list of kwargs, so before this an
# unrecognized top-level kwarg was dropped on the floor: the request succeeded but the schema
# contract silently became prompt compliance (#85626 review, point 2).
top_level_response_format = kwargs.get("response_format")
if top_level_response_format is not None:
_translate_anthropic_response_format(anthropic_kwargs, top_level_response_format)
# AnthropicBedrock exception: Bedrock's InvokeModel endpoint rejects the translated field
# with 400 ``output_config.format: Extra inputs are not permitted`` (ValidationException on
# InvokeModelWithResponseStream, every Claude model — endpoint, not model). The
# per-process _REJECTED_ROUTES memo only learns this after the first 400 and forgets it on
# every restart/gateway process, so skip the translation up front for that client; schema
# enforcement degrades to prompt compliance, the same fallback as an unsupported format
# (#124923). The Mantle route for Bedrock-hosted OpenAI models uses a plain OpenAI client
# and is not affected.
def _translate_or_skip(rf: Any) -> None:
if rf is None:
return
if self._is_bedrock_client:
logger.info(
"AnthropicBedrock route: omitting the structured-output format field "
"(Bedrock rejects output_config.format with 400); schema enforcement "
"degrades to prompt compliance",
)
return
_translate_anthropic_response_format(anthropic_kwargs, rf)

_translate_or_skip(kwargs.get("response_format"))
caller_extra_body = kwargs.get("extra_body")
if caller_extra_body and isinstance(caller_extra_body, dict):
_translate_anthropic_response_format(anthropic_kwargs, caller_extra_body.get("response_format"))
_translate_or_skip(caller_extra_body.get("response_format"))
passthrough = {
k: v for k, v in caller_extra_body.items()
if k not in {"reasoning", "response_format"} and not str(k).startswith("_")
Expand Down
6 changes: 5 additions & 1 deletion agent/bedrock_adapter.py
Original file line number Diff line number Diff line change
Expand Up @@ -1214,8 +1214,12 @@ def probe_bedrock_context_length(model_id: str, region: str) -> Optional[int]:
for tier_tokens in _BEDROCK_PROBE_TIERS:
oversized = "data " * int(tier_tokens / _WORDS_PER_TOKEN)
try:
# maxTokens must clear every model's minimum: OpenAI-on-Bedrock (gpt-6-sol/luna) rejects
# max output < 16 with a ValidationException BEFORE the prompt-length check, so the old
# 8 never produced the parseable "prompt is too long" error — the probe just burned two
# ~1.3M/2.2M-token requests per process and fell back to the 128K default (#124923).
client.converse(modelId=model_id, messages=[{"role": "user", "content": [{"text": oversized}]}],
inferenceConfig={"maxTokens": 8})
inferenceConfig={"maxTokens": 16})
logger.debug("Bedrock context probe for %s accepted ~%s-token prompt; "
"window is at least that", model_id, f"{tier_tokens:,}")
return tier_tokens
Expand Down
51 changes: 51 additions & 0 deletions tests/agent/test_auxiliary_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -3033,6 +3033,57 @@ def test_reasoning_config_reaches_native_anthropic_wire_kwargs(self):
assert captured["output_config"] == {"effort": "medium"}
assert "extra_body" not in captured

def test_bedrock_client_never_receives_output_config_format(self):
"""AnthropicBedrock's InvokeModel endpoint rejects the translated
output_config.format with 400 'Extra inputs are not permitted' (every
Claude model) and the per-process rejection memo forgets it on every
restart, so the adapter must skip the translation up front rather than
burn a 400 per structured-output call per process (#124923)."""
from agent.auxiliary_client import _AnthropicCompletionsAdapter

captured = {}

class _Messages:
def create(self, **kwargs):
captured.update(kwargs)
return SimpleNamespace(
content=[SimpleNamespace(type="text", text="ok")],
stop_reason="end_turn",
usage=SimpleNamespace(input_tokens=1, output_tokens=1, total_tokens=2),
)

bedrock_client = type("AnthropicBedrock", (), {"messages": _Messages()})()
adapter = _AnthropicCompletionsAdapter(bedrock_client, "anthropic.claude-opus-5-5", is_oauth=False)

response_format = {"type": "json_schema", "json_schema": {
"name": "session_title", "strict": True, "schema": {"type": "object"}}}
adapter.create(
model="anthropic.claude-opus-5-5",
messages=[{"role": "user", "content": "hi"}],
extra_body={"response_format": response_format},
)

# No output_config.format on the wire; the call itself succeeds and the
# passthrough extra_body no longer carries response_format either.
assert captured.get("output_config", {}).get("format") is None
assert "response_format" not in captured.get("extra_body", {})

def test_non_bedrock_client_still_gets_output_config_format(self):
"""The skip must be scoped to AnthropicBedrock: direct/partner Anthropic
and Messages-wire gateways accept output_config.format, and dropping it
there would silently degrade structured output to prompt compliance."""
adapter, captured = self._build_adapter()

adapter.create(
model="claude-fable-5",
messages=[{"role": "user", "content": "hi"}],
extra_body={"response_format": {"type": "json_schema", "json_schema": {
"name": "session_title", "strict": True, "schema": {"type": "object"}}}},
)

assert captured["output_config"]["format"] == {
"type": "json_schema", "schema": {"type": "object"}}

def test_build_call_kwargs_private_reasoning_only_for_anthropic_messages(self):
anthropic_kwargs = _build_call_kwargs(
"anthropic",
Expand Down
16 changes: 16 additions & 0 deletions tests/agent/test_bedrock_adapter.py
Original file line number Diff line number Diff line change
Expand Up @@ -1207,6 +1207,22 @@ def test_probe_result_beats_static_table(self):
"eu.anthropic.claude-opus-4-8",
region="eu-central-1") == 1_000_000

def test_probe_max_tokens_meets_openai_bedrock_minimum(self):
"""OpenAI-on-Bedrock (gpt-6-sol/luna) rejects max output < 16 with a
ValidationException BEFORE the prompt-length check, so the probe must
send at least 16 or it never sees the parseable 'prompt is too long'
error (#124923)."""
from agent.bedrock_adapter import probe_bedrock_context_length
client = MagicMock()
client.converse.side_effect = Exception(
"prompt is too long: 5000032 tokens > 272000 maximum")
with patch("agent.bedrock_adapter._get_bedrock_runtime_client",
return_value=client):
assert probe_bedrock_context_length(
"openai.gpt-6-sol", "us-east-1") == 272_000
sent = client.converse.call_args.kwargs["inferenceConfig"]
assert sent["maxTokens"] >= 16


# ---------------------------------------------------------------------------
# Tool-calling capability detection
Expand Down