From edab78f86c0f7da31db26d70dc6ef75a9115abc3 Mon Sep 17 00:00:00 2001 From: Yuan Li Date: Sun, 27 Sep 2026 15:46:51 +0800 Subject: [PATCH 1/2] fix(bedrock): stop sending output_config.format on AnthropicBedrock aux routes; probe with maxTokens=16 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit title_generation and other structured-output aux calls translate response_format to output_config.format even for AnthropicBedrock clients, but Bedrock's InvokeModel endpoint rejects that field with 400 'output_config.format: Extra inputs are not permitted' (every Claude model). The per-process rejection memo forgets this on every restart, so each new session's title call re-hits the 400 and falls through the retry ladder. Skip the translation up front for AnthropicBedrock clients; schema enforcement degrades to prompt compliance. The Bedrock context-length probe sent maxTokens=8, below the OpenAI-on-Bedrock minimum of 16: gpt-6-sol/luna raise a ValidationException (max_output_tokens below minimum) before the prompt-length check, so the probe never parsed a limit — it burned ~3.5M tokens of probe requests per process and resolved gpt-6 models to the 128K default, triggering early compression. Probe with 16. Fixes #124923 --- agent/auxiliary_client.py | 31 +++++++++++++++++++++++++++---- agent/bedrock_adapter.py | 6 +++++- 2 files changed, 32 insertions(+), 5 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 83c7fc162426d..69beb14f3414e 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -1703,6 +1703,11 @@ def __init__(self, real_client: Any, model: str, is_oauth: bool = False, base_ur self._client = real_client self._model = model self._is_oauth = is_oauth + # AnthropicBedrock detection by class name: the anthropic SDK is an optional extra, and + # this module must stay importable without it. Only Bedrock's InvokeModel endpoint rejects + # the translated output_config.format (#124923); direct/partner Anthropic and Messages-wire + # gateways accept it. + self._is_bedrock_client = type(real_client).__name__ == "AnthropicBedrock" # Caller URL first; fall back to the SDK client's host only for Nous Portal — a blanket # fallback would flip MiniMax/Zhipu aux adapters to third-party handling (strips thinking sigs). self._base_url = base_url or None @@ -1770,12 +1775,30 @@ def create(self, **kwargs) -> Any: # The adapter builds the Messages body from a fixed allow-list of kwargs, so before this an # unrecognized top-level kwarg was dropped on the floor: the request succeeded but the schema # contract silently became prompt compliance (#85626 review, point 2). - top_level_response_format = kwargs.get("response_format") - if top_level_response_format is not None: - _translate_anthropic_response_format(anthropic_kwargs, top_level_response_format) + # AnthropicBedrock exception: Bedrock's InvokeModel endpoint rejects the translated field + # with 400 ``output_config.format: Extra inputs are not permitted`` (ValidationException on + # InvokeModelWithResponseStream, every Claude model — endpoint, not model). The + # per-process _REJECTED_ROUTES memo only learns this after the first 400 and forgets it on + # every restart/gateway process, so skip the translation up front for that client; schema + # enforcement degrades to prompt compliance, the same fallback as an unsupported format + # (#124923). The Mantle route for Bedrock-hosted OpenAI models uses a plain OpenAI client + # and is not affected. + def _translate_or_skip(rf: Any) -> None: + if rf is None: + return + if self._is_bedrock_client: + logger.info( + "AnthropicBedrock route: omitting the structured-output format field " + "(Bedrock rejects output_config.format with 400); schema enforcement " + "degrades to prompt compliance", + ) + return + _translate_anthropic_response_format(anthropic_kwargs, rf) + + _translate_or_skip(kwargs.get("response_format")) caller_extra_body = kwargs.get("extra_body") if caller_extra_body and isinstance(caller_extra_body, dict): - _translate_anthropic_response_format(anthropic_kwargs, caller_extra_body.get("response_format")) + _translate_or_skip(caller_extra_body.get("response_format")) passthrough = { k: v for k, v in caller_extra_body.items() if k not in {"reasoning", "response_format"} and not str(k).startswith("_") diff --git a/agent/bedrock_adapter.py b/agent/bedrock_adapter.py index 555fe4f58dd38..d35de45692646 100644 --- a/agent/bedrock_adapter.py +++ b/agent/bedrock_adapter.py @@ -1214,8 +1214,12 @@ def probe_bedrock_context_length(model_id: str, region: str) -> Optional[int]: for tier_tokens in _BEDROCK_PROBE_TIERS: oversized = "data " * int(tier_tokens / _WORDS_PER_TOKEN) try: + # maxTokens must clear every model's minimum: OpenAI-on-Bedrock (gpt-6-sol/luna) rejects + # max output < 16 with a ValidationException BEFORE the prompt-length check, so the old + # 8 never produced the parseable "prompt is too long" error — the probe just burned two + # ~1.3M/2.2M-token requests per process and fell back to the 128K default (#124923). client.converse(modelId=model_id, messages=[{"role": "user", "content": [{"text": oversized}]}], - inferenceConfig={"maxTokens": 8}) + inferenceConfig={"maxTokens": 16}) logger.debug("Bedrock context probe for %s accepted ~%s-token prompt; " "window is at least that", model_id, f"{tier_tokens:,}") return tier_tokens From d496e17c7a58462cc5a2097bcb5cfa548ac9ce91 Mon Sep 17 00:00:00 2001 From: Yuan Li Date: Sun, 27 Sep 2026 15:50:32 +0800 Subject: [PATCH 2/2] test(bedrock): regression tests for the AnthropicBedrock format skip and the 16-token probe floor - adapter: AnthropicBedrock clients must never receive output_config.format (Bedrock 400 'Extra inputs are not permitted', re-learned every process); non-Bedrock Messages clients must still get the translation. - probe: inferenceConfig.maxTokens >= 16 so OpenAI-on-Bedrock models reach the prompt-length check instead of failing validation first. --- tests/agent/test_auxiliary_client.py | 51 ++++++++++++++++++++++++++++ tests/agent/test_bedrock_adapter.py | 16 +++++++++ 2 files changed, 67 insertions(+) diff --git a/tests/agent/test_auxiliary_client.py b/tests/agent/test_auxiliary_client.py index 9e2960ca07300..077e0581d01ee 100644 --- a/tests/agent/test_auxiliary_client.py +++ b/tests/agent/test_auxiliary_client.py @@ -3033,6 +3033,57 @@ def test_reasoning_config_reaches_native_anthropic_wire_kwargs(self): assert captured["output_config"] == {"effort": "medium"} assert "extra_body" not in captured + def test_bedrock_client_never_receives_output_config_format(self): + """AnthropicBedrock's InvokeModel endpoint rejects the translated + output_config.format with 400 'Extra inputs are not permitted' (every + Claude model) and the per-process rejection memo forgets it on every + restart, so the adapter must skip the translation up front rather than + burn a 400 per structured-output call per process (#124923).""" + from agent.auxiliary_client import _AnthropicCompletionsAdapter + + captured = {} + + class _Messages: + def create(self, **kwargs): + captured.update(kwargs) + return SimpleNamespace( + content=[SimpleNamespace(type="text", text="ok")], + stop_reason="end_turn", + usage=SimpleNamespace(input_tokens=1, output_tokens=1, total_tokens=2), + ) + + bedrock_client = type("AnthropicBedrock", (), {"messages": _Messages()})() + adapter = _AnthropicCompletionsAdapter(bedrock_client, "anthropic.claude-opus-5-5", is_oauth=False) + + response_format = {"type": "json_schema", "json_schema": { + "name": "session_title", "strict": True, "schema": {"type": "object"}}} + adapter.create( + model="anthropic.claude-opus-5-5", + messages=[{"role": "user", "content": "hi"}], + extra_body={"response_format": response_format}, + ) + + # No output_config.format on the wire; the call itself succeeds and the + # passthrough extra_body no longer carries response_format either. + assert captured.get("output_config", {}).get("format") is None + assert "response_format" not in captured.get("extra_body", {}) + + def test_non_bedrock_client_still_gets_output_config_format(self): + """The skip must be scoped to AnthropicBedrock: direct/partner Anthropic + and Messages-wire gateways accept output_config.format, and dropping it + there would silently degrade structured output to prompt compliance.""" + adapter, captured = self._build_adapter() + + adapter.create( + model="claude-fable-5", + messages=[{"role": "user", "content": "hi"}], + extra_body={"response_format": {"type": "json_schema", "json_schema": { + "name": "session_title", "strict": True, "schema": {"type": "object"}}}}, + ) + + assert captured["output_config"]["format"] == { + "type": "json_schema", "schema": {"type": "object"}} + def test_build_call_kwargs_private_reasoning_only_for_anthropic_messages(self): anthropic_kwargs = _build_call_kwargs( "anthropic", diff --git a/tests/agent/test_bedrock_adapter.py b/tests/agent/test_bedrock_adapter.py index 36ef2d072e4ec..c785a5befc9a7 100644 --- a/tests/agent/test_bedrock_adapter.py +++ b/tests/agent/test_bedrock_adapter.py @@ -1207,6 +1207,22 @@ def test_probe_result_beats_static_table(self): "eu.anthropic.claude-opus-4-8", region="eu-central-1") == 1_000_000 + def test_probe_max_tokens_meets_openai_bedrock_minimum(self): + """OpenAI-on-Bedrock (gpt-6-sol/luna) rejects max output < 16 with a + ValidationException BEFORE the prompt-length check, so the probe must + send at least 16 or it never sees the parseable 'prompt is too long' + error (#124923).""" + from agent.bedrock_adapter import probe_bedrock_context_length + client = MagicMock() + client.converse.side_effect = Exception( + "prompt is too long: 5000032 tokens > 272000 maximum") + with patch("agent.bedrock_adapter._get_bedrock_runtime_client", + return_value=client): + assert probe_bedrock_context_length( + "openai.gpt-6-sol", "us-east-1") == 272_000 + sent = client.converse.call_args.kwargs["inferenceConfig"] + assert sent["maxTokens"] >= 16 + # --------------------------------------------------------------------------- # Tool-calling capability detection