Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
27 changes: 21 additions & 6 deletions agent/auxiliary_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -1699,10 +1699,14 @@ def _translate_anthropic_response_format(anthropic_kwargs: Dict[str, Any], respo
class _AnthropicCompletionsAdapter:
"""OpenAI-client-compatible adapter for Anthropic Messages API."""

def __init__(self, real_client: Any, model: str, is_oauth: bool = False, base_url: str | None = None):
def __init__(
self, real_client: Any, model: str, is_oauth: bool = False,
base_url: str | None = None, is_bedrock: bool = False,
):
self._client = real_client
self._model = model
self._is_oauth = is_oauth
self._is_bedrock = is_bedrock
# Caller URL first; fall back to the SDK client's host only for Nous Portal — a blanket
# fallback would flip MiniMax/Zhipu aux adapters to third-party handling (strips thinking sigs).
self._base_url = base_url or None
Expand Down Expand Up @@ -1771,11 +1775,15 @@ def create(self, **kwargs) -> Any:
# unrecognized top-level kwarg was dropped on the floor: the request succeeded but the schema
# contract silently became prompt compliance (#85626 review, point 2).
top_level_response_format = kwargs.get("response_format")
if top_level_response_format is not None:
# AnthropicBedrock rejects output_config.format at the Bedrock endpoint (#124923).
# Keep native Anthropic / compatible Messages routes schema-enforced, but let Bedrock
# fall back to the same prompt-compliance behavior used after a structured-output 400.
if top_level_response_format is not None and not self._is_bedrock:
_translate_anthropic_response_format(anthropic_kwargs, top_level_response_format)
caller_extra_body = kwargs.get("extra_body")
if caller_extra_body and isinstance(caller_extra_body, dict):
_translate_anthropic_response_format(anthropic_kwargs, caller_extra_body.get("response_format"))
if not self._is_bedrock:
_translate_anthropic_response_format(anthropic_kwargs, caller_extra_body.get("response_format"))
passthrough = {
k: v for k, v in caller_extra_body.items()
if k not in {"reasoning", "response_format"} and not str(k).startswith("_")
Expand Down Expand Up @@ -1814,9 +1822,14 @@ def create(self, **kwargs) -> Any:
class AnthropicAuxiliaryClient:
"""OpenAI-client-compatible wrapper over a native Anthropic client."""

def __init__(self, real_client: Any, model: str, api_key: str, base_url: str, is_oauth: bool = False):
def __init__(
self, real_client: Any, model: str, api_key: str, base_url: str,
is_oauth: bool = False, is_bedrock: bool = False,
):
self._real_client = real_client
self.chat = _ChatShim(_AnthropicCompletionsAdapter(real_client, model, is_oauth=is_oauth, base_url=base_url))
self.chat = _ChatShim(_AnthropicCompletionsAdapter(
real_client, model, is_oauth=is_oauth, base_url=base_url, is_bedrock=is_bedrock,
))
self.api_key = api_key
self.base_url = base_url

Expand Down Expand Up @@ -4817,7 +4830,9 @@ def _build_bedrock_client(provider: str, model: Optional[str], *, raw_codex: boo
except ImportError as exc:
logger.warning("resolve_provider_client: cannot create Bedrock client: %s", exc)
return None, None
client = AnthropicAuxiliaryClient(real_client, final_model, api_key="aws-sdk", base_url=base_url)
client = AnthropicAuxiliaryClient(
real_client, final_model, api_key="aws-sdk", base_url=base_url, is_bedrock=True,
)
logger.debug("resolve_provider_client: bedrock anthropic (%s, %s)", final_model, region)
else:
client = BedrockAuxiliaryClient(region, final_model)
Expand Down
4 changes: 3 additions & 1 deletion agent/bedrock_adapter.py
Original file line number Diff line number Diff line change
Expand Up @@ -1214,8 +1214,10 @@ def probe_bedrock_context_length(model_id: str, region: str) -> Optional[int]:
for tier_tokens in _BEDROCK_PROBE_TIERS:
oversized = "data " * int(tier_tokens / _WORDS_PER_TOKEN)
try:
# OpenAI-on-Bedrock models reject output caps below 16 before context-length
# validation, which masks the very error this probe needs to parse (#124923).
client.converse(modelId=model_id, messages=[{"role": "user", "content": [{"text": oversized}]}],
inferenceConfig={"maxTokens": 8})
inferenceConfig={"maxTokens": 16})
logger.debug("Bedrock context probe for %s accepted ~%s-token prompt; "
"window is at least that", model_id, f"{tier_tokens:,}")
return tier_tokens
Expand Down
86 changes: 86 additions & 0 deletions tests/agent/test_bedrock_issue_124923.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,86 @@
"""Regression coverage for Bedrock auxiliary request shaping (#124923)."""

from types import SimpleNamespace
from unittest.mock import MagicMock, patch


_RESPONSE_FORMAT = {
"type": "json_schema",
"json_schema": {
"name": "thread_title",
"schema": {
"type": "object",
"properties": {"title": {"type": "string"}},
"required": ["title"],
},
},
}


def _capture_anthropic_kwargs(*, is_bedrock: bool) -> dict:
from agent.auxiliary_client import _AnthropicCompletionsAdapter

captured = {}
adapter = _AnthropicCompletionsAdapter(
MagicMock(name="anthropic_client"),
"global.anthropic.claude-sonnet-5",
base_url="https://bedrock-runtime.us-east-1.amazonaws.com",
is_bedrock=is_bedrock,
)

def _fake_create(_client, api_kwargs, **_kwargs):
captured.update(api_kwargs)
return SimpleNamespace()

normalized = SimpleNamespace(
content="ok", tool_calls=None, reasoning=None, finish_reason="stop",
)
with patch(
"agent.anthropic_adapter.create_anthropic_message",
side_effect=_fake_create,
), patch("agent.transports.get_transport") as mock_get_transport:
mock_get_transport.return_value.normalize_response.return_value = normalized
adapter.create(
model="global.anthropic.claude-sonnet-5",
messages=[{"role": "user", "content": "title this"}],
max_tokens=64,
extra_body={"response_format": _RESPONSE_FORMAT},
)
return captured


def test_anthropic_bedrock_omits_structured_output_format():
kwargs = _capture_anthropic_kwargs(is_bedrock=True)

assert "format" not in (kwargs.get("output_config") or {})
assert "response_format" not in kwargs
assert "response_format" not in (kwargs.get("extra_body") or {})


def test_native_anthropic_still_translates_structured_output_format():
kwargs = _capture_anthropic_kwargs(is_bedrock=False)

assert kwargs["output_config"]["format"]["type"] == "json_schema"
assert kwargs["output_config"]["format"]["schema"] == _RESPONSE_FORMAT["json_schema"]["schema"]


def test_context_probe_uses_bedrock_safe_minimum_output(monkeypatch):
from agent import bedrock_adapter

class _ProbeClient:
def __init__(self):
self.inference_configs = []

def converse(self, **kwargs):
self.inference_configs.append(kwargs["inferenceConfig"])
raise RuntimeError("prompt is too long: 1300 tokens > 1200 maximum")

client = _ProbeClient()
monkeypatch.setattr(bedrock_adapter, "_get_bedrock_runtime_client", lambda _region: client)
monkeypatch.setattr(bedrock_adapter, "_BEDROCK_PROBE_TIERS", (1300,))
monkeypatch.setattr(bedrock_adapter, "_WORDS_PER_TOKEN", 1.0)

assert bedrock_adapter.probe_bedrock_context_length(
"openai.gpt-5.6-sol", "us-east-1",
) == 1200
assert client.inference_configs == [{"maxTokens": 16}]