From 77c3dadd565a860a075ff376a94c4daff23448de Mon Sep 17 00:00:00 2001 From: fangliquanflq Date: Thu, 17 Sep 2026 01:08:19 +0800 Subject: [PATCH 1/2] fix(auxiliary): select supported structured output format --- agent/auxiliary_client.py | 22 ++++++++++++++- agent/models_dev.py | 28 +++++++++++++++++-- tests/agent/test_models_dev.py | 13 +++++++++ .../test_structured_output_rejection_retry.py | 15 ++++++++++ 4 files changed, 75 insertions(+), 3 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index b01b764098c55..b938dc307d026 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -6168,6 +6168,24 @@ def _merge_aux_extra_body( return merged_extra +def _select_structured_output_format(provider: str, model: str, extra_body: Dict[str, Any]) -> Dict[str, Any]: + """Downgrade ``json_schema`` to JSON mode unless the resolved model supports it.""" + response_format = extra_body.get("response_format") + if not isinstance(response_format, dict) or response_format.get("type") != "json_schema": + return extra_body + try: + from agent.models_dev import model_supports_json_schema + supports_json_schema = model_supports_json_schema(provider, model) + except Exception as exc: + logger.debug("Structured-output capability lookup failed for %s/%s: %s", provider, model, exc) + supports_json_schema = False + if supports_json_schema: + return extra_body + selected = dict(extra_body) + selected["response_format"] = {"type": "json_object"} + return selected + + def _build_call_kwargs( provider: str, model: str, messages: list, temperature: Optional[float] = None, max_tokens: Optional[int] = None, tools: Optional[list] = None, timeout: float = 30.0, @@ -6198,7 +6216,9 @@ def _build_call_kwargs( # ``extra_body.reasoning`` fallback. projection = _project_provider_profile(provider, provider_norm, model, effective_base, reasoning_config) kwargs.update(projection.top_level) - if merged_extra := _merge_aux_extra_body(extra_body, projection, reasoning_config, provider_norm): + merged_extra = _merge_aux_extra_body(extra_body, projection, reasoning_config, provider_norm) + merged_extra = _select_structured_output_format(provider, model, merged_extra) + if merged_extra: kwargs["extra_body"] = merged_extra # Anthropic Messages adapters take reasoning via a private kwarg that plain OpenAI SDK clients # would reject; Portal Claude is dual-wire, so include it only when the catalog id selects diff --git a/agent/models_dev.py b/agent/models_dev.py index fd40784b9b708..fd254111d3709 100644 --- a/agent/models_dev.py +++ b/agent/models_dev.py @@ -557,7 +557,7 @@ def lookup_models_dev_context(provider: str, model: str, *, allow_network: bool # Per-model overrides (config.yaml → model_overrides). Canonical schema (the ONLY key space consumers # accept): context_window, supports_tools, supports_vision, supports_reasoning, -# model_family. ``.`` is an explicit partial patch that always wins over the +# supports_structured_output, model_family. ``.`` is an explicit partial patch that always wins over the # catalog. ``._default`` / top-level ``_default`` are FILL-GAP defaults: they apply ONLY to # models the catalog does not know and never displace catalog data. Provider keys accept the Hermes # or models.dev id; model ids match exactly, then case-insensitively (mirroring catalog lookup). @@ -684,7 +684,11 @@ def _override_to_catalog_shape(override: Dict[str, Any]) -> Tuple[Dict[str, Any] } if limit: patch["limit"] = limit - for override_key, catalog_key in (("supports_tools", "tool_call"), ("supports_reasoning", "reasoning")): + for override_key, catalog_key in ( + ("supports_tools", "tool_call"), + ("supports_reasoning", "reasoning"), + ("supports_structured_output", "structured_output"), + ): if override_key in override: patch[catalog_key] = bool(override[override_key]) vision: Optional[bool] = None @@ -852,3 +856,23 @@ def get_model_info(provider_id: str, model_id: str, *, allow_network: bool = Fal # Not in catalog — an override (explicit or _default) may still provide it. raw = _apply_overrides(provider_id, model_id, entry) return _parse_model_info(mid, raw, mdev_id) if raw is not None else None + + +def model_supports_json_schema(provider_id: str, model_id: str) -> bool: + """Whether an auxiliary request may send OpenAI ``json_schema`` response format. + + An explicit per-model override is authoritative. DeepSeek's native API only implements + ``json_object`` even when catalog metadata describes the model as supporting structured + output, so its provider default is deliberately narrower. Other models must be positively + identified by the local catalog; unknown models use the broadly compatible JSON-object mode. + """ + override = _explicit_model_override(provider_id, model_id) + if override is not None and "supports_structured_output" in override: + return bool(override["supports_structured_output"]) + provider_key = PROVIDER_TO_MODELS_DEV.get( + (provider_id or "").strip(), (provider_id or "").strip() + ).lower() + if provider_key == "deepseek": + return False + info = get_model_info(provider_id, model_id, allow_network=False) + return bool(info and info.structured_output) diff --git a/tests/agent/test_models_dev.py b/tests/agent/test_models_dev.py index 180fdac845601..13e824334a6b5 100644 --- a/tests/agent/test_models_dev.py +++ b/tests/agent/test_models_dev.py @@ -20,6 +20,7 @@ get_model_info, get_provider_info, lookup_models_dev_context, + model_supports_json_schema, ) @@ -1063,6 +1064,18 @@ def test_caps_override_unknown_model(self): assert caps.supports_reasoning is False assert caps.supports_tools is True + def test_structured_output_override_controls_json_schema_capability(self): + overrides = { + "deepseek": { + "deepseek-flash": {"supports_structured_output": True}, + }, + } + with self._setup_overrides(overrides): + assert model_supports_json_schema("deepseek", "deepseek-flash") is True + + with self._setup_overrides({}): + assert model_supports_json_schema("deepseek", "deepseek-flash") is False + def test_caps_override_patches_existing_catalog_entry(self): """Explicit override patches specific fields on a known entry (#84482).""" overrides = { diff --git a/tests/agent/test_structured_output_rejection_retry.py b/tests/agent/test_structured_output_rejection_retry.py index 397f40c998422..9187e701695f5 100644 --- a/tests/agent/test_structured_output_rejection_retry.py +++ b/tests/agent/test_structured_output_rejection_retry.py @@ -33,6 +33,7 @@ from agent.auxiliary_client import ( call_llm, async_call_llm, + _build_call_kwargs, _is_structured_output_rejection, _without_structured_output_format, ) @@ -247,6 +248,20 @@ def test_no_retry_when_no_response_format_was_sent(self): assert client.chat.completions.create.call_count == 1 +def test_deepseek_uses_json_object_before_dispatch(): + extra_body = {"response_format": dict(_TITLE_RESPONSE_FORMAT)} + + kwargs = _build_call_kwargs( + "deepseek", + "deepseek-flash", + [{"role": "user", "content": "hi"}], + extra_body=extra_body, + ) + + assert kwargs["extra_body"]["response_format"] == {"type": "json_object"} + assert extra_body["response_format"]["type"] == "json_schema" + + class TestAsyncCallLlmStructuredOutputRetry: """``async_call_llm`` mirror of the sync retry semantics.""" From cf94bac62bdb06f8ac1fea08953253a53d022e9c Mon Sep 17 00:00:00 2001 From: fangliquanflq Date: Thu, 17 Sep 2026 01:21:38 +0800 Subject: [PATCH 2/2] test(auxiliary): preserve supported JSON schemas --- hermes_cli/config_defaults.py | 3 ++- .../test_structured_output_rejection_retry.py | 25 +++++++++++++++++++ 2 files changed, 27 insertions(+), 1 deletion(-) diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 0c8a6f363ca51..ff88399ffe18a 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1887,7 +1887,8 @@ def _aux(timeout, *, reasoning_effort=True, **extra): "providers": {}, }, # Per-model metadata overrides. Fields: context_window, supports_tools, - # supports_vision, supports_reasoning, model_family. . wins over + # supports_vision, supports_reasoning, supports_structured_output, model_family. + # . wins over # models.dev/OpenRouter/hardcoded defaults for the fields it sets (chain order in # agent/model_metadata.py). ._default and top-level _default fill gaps ONLY for models # the catalog does not know, so they never clamp known models. Unknown ids start from safe diff --git a/tests/agent/test_structured_output_rejection_retry.py b/tests/agent/test_structured_output_rejection_retry.py index 9187e701695f5..39d59dea4e8cd 100644 --- a/tests/agent/test_structured_output_rejection_retry.py +++ b/tests/agent/test_structured_output_rejection_retry.py @@ -247,6 +247,31 @@ def test_no_retry_when_no_response_format_was_sent(self): ) assert client.chat.completions.create.call_count == 1 + def test_capable_model_dispatches_complete_json_schema(self): + client = MagicMock() + client.base_url = "https://api.openai.com/v1" + client.chat.completions.create.return_value = _dummy_response() + + with ( + patch("agent.auxiliary_client._resolve_task_provider_model", + return_value=("openai", "schema-model", None, None, None)), + patch("agent.auxiliary_client._get_cached_client", + return_value=(client, "schema-model")), + patch("agent.models_dev.model_supports_json_schema", return_value=True), + patch("agent.auxiliary_client._validate_llm_response", + side_effect=lambda resp, _task, **_kw: resp), + ): + result = call_llm( + task="title_generation", + messages=[{"role": "user", "content": "hi"}], + max_tokens=64, + extra_body={"response_format": dict(_TITLE_RESPONSE_FORMAT)}, + ) + + assert result == {"ok": True} + dispatched = client.chat.completions.create.call_args.kwargs + assert dispatched["extra_body"]["response_format"] == _TITLE_RESPONSE_FORMAT + def test_deepseek_uses_json_object_before_dispatch(): extra_body = {"response_format": dict(_TITLE_RESPONSE_FORMAT)}