Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
22 changes: 21 additions & 1 deletion agent/auxiliary_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -6168,6 +6168,24 @@ def _merge_aux_extra_body(
return merged_extra


def _select_structured_output_format(provider: str, model: str, extra_body: Dict[str, Any]) -> Dict[str, Any]:
"""Downgrade ``json_schema`` to JSON mode unless the resolved model supports it."""
response_format = extra_body.get("response_format")
if not isinstance(response_format, dict) or response_format.get("type") != "json_schema":
return extra_body
try:
from agent.models_dev import model_supports_json_schema
supports_json_schema = model_supports_json_schema(provider, model)
except Exception as exc:
logger.debug("Structured-output capability lookup failed for %s/%s: %s", provider, model, exc)
supports_json_schema = False
if supports_json_schema:
return extra_body
selected = dict(extra_body)
selected["response_format"] = {"type": "json_object"}
return selected


def _build_call_kwargs(
provider: str, model: str, messages: list, temperature: Optional[float] = None,
max_tokens: Optional[int] = None, tools: Optional[list] = None, timeout: float = 30.0,
Expand Down Expand Up @@ -6198,7 +6216,9 @@ def _build_call_kwargs(
# ``extra_body.reasoning`` fallback.
projection = _project_provider_profile(provider, provider_norm, model, effective_base, reasoning_config)
kwargs.update(projection.top_level)
if merged_extra := _merge_aux_extra_body(extra_body, projection, reasoning_config, provider_norm):
merged_extra = _merge_aux_extra_body(extra_body, projection, reasoning_config, provider_norm)
merged_extra = _select_structured_output_format(provider, model, merged_extra)
if merged_extra:
kwargs["extra_body"] = merged_extra
# Anthropic Messages adapters take reasoning via a private kwarg that plain OpenAI SDK clients
# would reject; Portal Claude is dual-wire, so include it only when the catalog id selects
Expand Down
28 changes: 26 additions & 2 deletions agent/models_dev.py
Original file line number Diff line number Diff line change
Expand Up @@ -557,7 +557,7 @@ def lookup_models_dev_context(provider: str, model: str, *, allow_network: bool

# Per-model overrides (config.yaml → model_overrides). Canonical schema (the ONLY key space consumers
# accept): context_window, supports_tools, supports_vision, supports_reasoning,
# model_family. ``<provider>.<model_id>`` is an explicit partial patch that always wins over the
# supports_structured_output, model_family. ``<provider>.<model_id>`` is an explicit partial patch that always wins over the
# catalog. ``<provider>._default`` / top-level ``_default`` are FILL-GAP defaults: they apply ONLY to
# models the catalog does not know and never displace catalog data. Provider keys accept the Hermes
# or models.dev id; model ids match exactly, then case-insensitively (mirroring catalog lookup).
Expand Down Expand Up @@ -684,7 +684,11 @@ def _override_to_catalog_shape(override: Dict[str, Any]) -> Tuple[Dict[str, Any]
}
if limit:
patch["limit"] = limit
for override_key, catalog_key in (("supports_tools", "tool_call"), ("supports_reasoning", "reasoning")):
for override_key, catalog_key in (
("supports_tools", "tool_call"),
("supports_reasoning", "reasoning"),
("supports_structured_output", "structured_output"),
):
if override_key in override:
patch[catalog_key] = bool(override[override_key])
vision: Optional[bool] = None
Expand Down Expand Up @@ -852,3 +856,23 @@ def get_model_info(provider_id: str, model_id: str, *, allow_network: bool = Fal
# Not in catalog — an override (explicit or _default) may still provide it.
raw = _apply_overrides(provider_id, model_id, entry)
return _parse_model_info(mid, raw, mdev_id) if raw is not None else None


def model_supports_json_schema(provider_id: str, model_id: str) -> bool:
"""Whether an auxiliary request may send OpenAI ``json_schema`` response format.

An explicit per-model override is authoritative. DeepSeek's native API only implements
``json_object`` even when catalog metadata describes the model as supporting structured
output, so its provider default is deliberately narrower. Other models must be positively
identified by the local catalog; unknown models use the broadly compatible JSON-object mode.
"""
override = _explicit_model_override(provider_id, model_id)
if override is not None and "supports_structured_output" in override:
return bool(override["supports_structured_output"])
provider_key = PROVIDER_TO_MODELS_DEV.get(
(provider_id or "").strip(), (provider_id or "").strip()
).lower()
if provider_key == "deepseek":
return False
info = get_model_info(provider_id, model_id, allow_network=False)
return bool(info and info.structured_output)
3 changes: 2 additions & 1 deletion hermes_cli/config_defaults.py
Original file line number Diff line number Diff line change
Expand Up @@ -1887,7 +1887,8 @@ def _aux(timeout, *, reasoning_effort=True, **extra):
"providers": {},
},
# Per-model metadata overrides. Fields: context_window, supports_tools,
# supports_vision, supports_reasoning, model_family. <provider>.<model_id> wins over
# supports_vision, supports_reasoning, supports_structured_output, model_family.
# <provider>.<model_id> wins over
# models.dev/OpenRouter/hardcoded defaults for the fields it sets (chain order in
# agent/model_metadata.py). <provider>._default and top-level _default fill gaps ONLY for models
# the catalog does not know, so they never clamp known models. Unknown ids start from safe
Expand Down
13 changes: 13 additions & 0 deletions tests/agent/test_models_dev.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@
get_model_info,
get_provider_info,
lookup_models_dev_context,
model_supports_json_schema,
)


Expand Down Expand Up @@ -1063,6 +1064,18 @@ def test_caps_override_unknown_model(self):
assert caps.supports_reasoning is False
assert caps.supports_tools is True

def test_structured_output_override_controls_json_schema_capability(self):
overrides = {
"deepseek": {
"deepseek-flash": {"supports_structured_output": True},
},
}
with self._setup_overrides(overrides):
assert model_supports_json_schema("deepseek", "deepseek-flash") is True

with self._setup_overrides({}):
assert model_supports_json_schema("deepseek", "deepseek-flash") is False

def test_caps_override_patches_existing_catalog_entry(self):
"""Explicit override patches specific fields on a known entry (#84482)."""
overrides = {
Expand Down
40 changes: 40 additions & 0 deletions tests/agent/test_structured_output_rejection_retry.py
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,7 @@
from agent.auxiliary_client import (
call_llm,
async_call_llm,
_build_call_kwargs,
_is_structured_output_rejection,
_without_structured_output_format,
)
Expand Down Expand Up @@ -246,6 +247,45 @@ def test_no_retry_when_no_response_format_was_sent(self):
)
assert client.chat.completions.create.call_count == 1

def test_capable_model_dispatches_complete_json_schema(self):
client = MagicMock()
client.base_url = "https://api.openai.com/v1"
client.chat.completions.create.return_value = _dummy_response()

with (
patch("agent.auxiliary_client._resolve_task_provider_model",
return_value=("openai", "schema-model", None, None, None)),
patch("agent.auxiliary_client._get_cached_client",
return_value=(client, "schema-model")),
patch("agent.models_dev.model_supports_json_schema", return_value=True),
patch("agent.auxiliary_client._validate_llm_response",
side_effect=lambda resp, _task, **_kw: resp),
):
result = call_llm(
task="title_generation",
messages=[{"role": "user", "content": "hi"}],
max_tokens=64,
extra_body={"response_format": dict(_TITLE_RESPONSE_FORMAT)},
)

assert result == {"ok": True}
dispatched = client.chat.completions.create.call_args.kwargs
assert dispatched["extra_body"]["response_format"] == _TITLE_RESPONSE_FORMAT


def test_deepseek_uses_json_object_before_dispatch():
extra_body = {"response_format": dict(_TITLE_RESPONSE_FORMAT)}

kwargs = _build_call_kwargs(
"deepseek",
"deepseek-flash",
[{"role": "user", "content": "hi"}],
extra_body=extra_body,
)

assert kwargs["extra_body"]["response_format"] == {"type": "json_object"}
assert extra_body["response_format"]["type"] == "json_schema"


class TestAsyncCallLlmStructuredOutputRetry:
"""``async_call_llm`` mirror of the sync retry semantics."""
Expand Down