Skip to content
132 changes: 80 additions & 52 deletions agent/auxiliary_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -3187,6 +3187,8 @@ def _is_unsupported_parameter_error(exc: Exception, param: str) -> bool:
"unsupported parameter", "unsupported_parameter", "not supported", "does not support",
"doesn't support", "is deprecated for this model",
"unknown parameter", "unrecognized request argument", "unrecognized parameter", "invalid parameter",
# Strict pydantic-validated gateways (Fireworks) name the unknown field this way (#109774).
"extra inputs are not permitted",
))


Expand Down Expand Up @@ -3885,8 +3887,13 @@ def _send(client: Any, request_kwargs: Dict[str, Any], dest: _FallbackDestinatio
),
task,
)
from agent.auxiliary_fallback_recovery import send_with_parameter_rungs

def _send_recovering(client: Any, request_kwargs: Dict[str, Any], dest: _FallbackDestination) -> Any:
return send_with_parameter_rungs(
lambda c, kw: _send(c, kw, dest), client, request_kwargs, task=task)
try:
return _send(fb_client, fb_kwargs, destination)
return _send_recovering(fb_client, fb_kwargs, destination)
except Exception as fb_err:
if not _is_auth_error(fb_err):
raise
Expand All @@ -3896,7 +3903,7 @@ def _send(client: Any, request_kwargs: Dict[str, Any], dest: _FallbackDestinatio
if retry is not None:
failed_destination = retry[2]
try:
return _send(*retry)
return _send_recovering(*retry)
except Exception as retry_err:
if not _is_auth_error(retry_err):
raise
Expand Down Expand Up @@ -3924,8 +3931,13 @@ async def _send(client: Any, request_kwargs: Dict[str, Any], dest: _FallbackDest
await _relay_async_completion(client, request_kwargs, provider=dest.provider, api_mode=dest.api_mode),
task,
)
from agent.auxiliary_fallback_recovery import send_with_parameter_rungs_async

async def _send_recovering(client: Any, request_kwargs: Dict[str, Any], dest: _FallbackDestination) -> Any:
return await send_with_parameter_rungs_async(
lambda c, kw: _send(c, kw, dest), client, request_kwargs, task=task)
try:
return await _send(fb_client, fb_kwargs, destination)
return await _send_recovering(fb_client, fb_kwargs, destination)
except Exception as fb_err:
if not _is_auth_error(fb_err):
raise
Expand All @@ -3935,7 +3947,7 @@ async def _send(client: Any, request_kwargs: Dict[str, Any], dest: _FallbackDest
if retry is not None:
failed_destination = retry[2]
try:
return await _send(*retry)
return await _send_recovering(*retry)
except Exception as retry_err:
if not _is_auth_error(retry_err):
raise
Expand Down Expand Up @@ -7007,11 +7019,16 @@ def _rung(step: "_LadderStep", accept: Callable[[Exception], bool]):

def _param_rung_accepts(exc: Exception) -> bool:
"""After a parameter-strip retry: fall through to the max_tokens/payment/auth
chains with the stripped kwargs; re-raise anything those chains won't handle."""
chains with the stripped kwargs; re-raise anything those chains won't handle.
A 429 on the retry is the credential/provider-fallback rungs' job, so it falls
through too (the pre-ladder max_tokens rung accepted rate limits)."""
return (_is_payment_error(exc) or _is_connection_error(exc) or _is_auth_error(exc)
or _is_rate_limit_error(exc)
or "max_tokens" in str(exc) or "unsupported_parameter" in str(exc)
# Parameter rungs chain (temperature-strip retry 400s on reasoning_effort / response_format),
# and a route-gating 400 after a strip still reaches the provider-fallback rung.
# Parameter rungs chain in any order (a reasoning-strip retry can 400 on temperature,
# a temperature-strip retry on max_tokens), and a route-gating 400 after a strip still
# reaches the provider-fallback rung.
or _is_unsupported_parameter_error(exc, "temperature")
or _is_reasoning_field_rejection(exc) or _is_structured_output_rejection(exc)
or _is_model_incompatible_error(exc))

Expand All @@ -7030,61 +7047,72 @@ def _credential_rung_accepts(exc: Exception) -> bool:
])


def _without_temperature(kwargs: dict) -> Optional[dict]:
"""Copy *kwargs* without ``temperature``; None when it was not sent."""
return {k: v for k, v in kwargs.items() if k != "temperature"} if "temperature" in kwargs else None


def _without_max_tokens(kwargs: dict) -> dict:
"""Copy *kwargs* without either output cap. Unlike the other strips this never returns None: a
route can translate the caller's cap into a field the wire kwargs no longer show (Codex
Responses), and the provider's own gateway may inject one — the 400 still names ``max_tokens``
and the identical request completes on retry (registry class #89897/#90257)."""
return {k: v for k, v in kwargs.items() if k not in ("max_tokens", "max_completion_tokens")}


def _is_max_tokens_rejection(exc: Exception, client: Any) -> bool:
err_str = str(exc)
# ZAI vision models reject max_tokens with code 1210 and a message that never
# mentions "max_tokens", so detect it explicitly.
is_zai_param_error = "1210" in err_str and "bigmodel" in str(getattr(client, "base_url", ""))
return ("max_tokens" in err_str or "unsupported_parameter" in err_str
or _is_unsupported_parameter_error(exc, "max_tokens") or is_zai_param_error)


def _parameter_rungs(client: Any, max_tokens: Optional[int]) -> tuple:
"""Ordered ``(matches, strip, log message)`` parameter rungs; ``strip`` returns None when the
field was not on the wire, so an unchanged request is never re-sent."""
return (
(lambda exc: _is_unsupported_parameter_error(exc, "temperature"), _without_temperature,
"provider rejected temperature; retrying without it"),
(_is_structured_output_rejection, _without_structured_output_format,
"provider rejected the structured-output format field; retrying without it "
"(schema enforcement degrades to prompt compliance)"),
# A chat-only model on an OpenAI-compatible relay rejects the profile's thinking-off encoding
# (top-level ``reasoning_effort: none``), and strict-schema gateways reject the generic
# ``extra_body.reasoning`` fallback outright (#109774); the caller only wanted "no thinking",
# so retry with every reasoning field omitted and let the route default apply (#112781).
(_is_reasoning_field_rejection, _without_reasoning_fields,
"provider rejected the reasoning field; retrying without it (route default applies)"),
(lambda exc: max_tokens is not None and _is_max_tokens_rejection(exc, client), _without_max_tokens,
"provider rejected the output cap; retrying without it"),
)


def _ladder_parameter_rungs(
first_err: Exception, route: _LadderRoute, kwargs: Dict[str, Any], max_tokens: Optional[int],
):
"""Rungs 1-4: retry without temperature / structured-output format / reasoning field / max_tokens.
"""Parameter rungs: retry without temperature / structured-output format / reasoning field /
max_tokens. Rungs chain in whichever order the provider raises them (reasoning models reject
temperature AND max_tokens; a reasoning-strip retry can then trip temperature, #78273), each
field stripped at most once, so a request with N rejected fields recovers in N retries.
Returns ``(response, None, kwargs)`` or ``(None, narrowed_err, stripped_kwargs)``."""
client, task, tag = route.client, route.task, route.tag
if "temperature" in kwargs and _is_unsupported_parameter_error(first_err, "temperature"):
retry_kwargs = {k: v for k, v in kwargs.items() if k != "temperature"}
logger.info("Auxiliary %s%s: provider rejected temperature; retrying once without it",
task or "call", tag)
rungs = list(_parameter_rungs(client, max_tokens))
while rungs:
hit = next(((matches, strip, message) for matches, strip, message in rungs
if matches(first_err) and strip(kwargs) is not None), None)
if hit is None:
break
rungs.remove(hit)
matches, strip, message = hit
retry_kwargs = strip(kwargs)
logger.info("Auxiliary %s%s: %s: %s", task or "call", tag, message, first_err)
resp, first_err = yield from _rung(
_LadderStep("call", (client, retry_kwargs)), _param_rung_accepts)
if first_err is None:
return resp, None, retry_kwargs
kwargs = retry_kwargs
if _is_structured_output_rejection(first_err):
retry_kwargs = _without_structured_output_format(kwargs)
if retry_kwargs is not None:
logger.info("Auxiliary %s%s: provider rejected the structured-output "
"format field; retrying once without it (schema "
"enforcement degrades to prompt compliance): %s", task or "call", tag, first_err)
resp, first_err = yield from _rung(
_LadderStep("call", (client, retry_kwargs)), _param_rung_accepts)
if first_err is None:
return resp, None, retry_kwargs
kwargs = retry_kwargs
# A chat-only model on an OpenAI-compatible relay rejects the profile's thinking-off encoding
# (top-level ``reasoning_effort: none``); the caller only wanted "no thinking", which is what
# such a model does anyway, so retry once with every reasoning field omitted (#112781).
if _is_reasoning_field_rejection(first_err):
retry_kwargs = _without_reasoning_fields(kwargs)
if retry_kwargs is not None:
logger.info("Auxiliary %s%s: provider rejected the reasoning field; retrying once "
"without it (route default applies): %s", task or "call", tag, first_err)
resp, first_err = yield from _rung(
_LadderStep("call", (client, retry_kwargs)), _param_rung_accepts)
if first_err is None:
return resp, None, retry_kwargs
kwargs = retry_kwargs
err_str = str(first_err)
# ZAI vision models reject max_tokens with code 1210 and a message that never
# mentions "max_tokens", so detect it explicitly.
_is_zai_param_error = "1210" in err_str and "bigmodel" in str(getattr(client, "base_url", ""))
if max_tokens is not None and (
"max_tokens" in err_str or "unsupported_parameter" in err_str
or _is_unsupported_parameter_error(first_err, "max_tokens") or _is_zai_param_error
):
kwargs.pop("max_tokens", None)
kwargs.pop("max_completion_tokens", None)
resp, first_err = yield from _rung(
_LadderStep("call", (client, kwargs)),
lambda exc: _is_payment_error(exc) or _is_connection_error(exc) or _is_rate_limit_error(exc),
)
if first_err is None:
return resp, None, kwargs
return None, first_err, kwargs


Expand Down
50 changes: 50 additions & 0 deletions agent/auxiliary_fallback_recovery.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,50 @@
"""Parameter-rejection recovery for auxiliary fallback candidates.

The primary auxiliary request runs the full recovery ladder in ``auxiliary_client``; a fallback
candidate (per-task chain, main fallback chain, discovery) used to get a single shot, so a fallback
model that rejects ``temperature``, ``max_tokens`` or a reasoning field failed the whole task even
though the same parameter rungs would have recovered it on the primary path (#78273, #72351).
This module runs those rungs — and only those — around a candidate's request.
"""
from typing import Any, Awaitable, Callable, Dict, Optional


def _parameter_ladder(first_err: Exception, client: Any, kwargs: Dict[str, Any], *,
task: Optional[str], tag: str):
from agent.auxiliary_client import _LadderRoute, _ladder_parameter_rungs
# Keyword construction: the route tuple grows with every new ladder rung (a positional 13-tuple
# broke the moment a sibling PR added ``timeout``); fields this ladder never reads stay None.
route = _LadderRoute(**{**dict.fromkeys(_LadderRoute._fields), "client": client, "task": task,
"tag": tag, "async_mode": bool(tag), "resolved_provider": "",
"base_info": str(getattr(client, "base_url", "") or "")})
max_tokens = kwargs.get("max_tokens") or kwargs.get("max_completion_tokens")
resp, err, _ = yield from _ladder_parameter_rungs(first_err, route, kwargs, max_tokens)
if err is None:
return resp
raise err


def send_with_parameter_rungs(
send: Callable[[Any, Dict[str, Any]], Any], client: Any, kwargs: Dict[str, Any], *, task: Optional[str],
) -> Any:
"""``send(client, kwargs)``; on a parameter 400, retry through the parameter rungs. Any other
error (auth, payment, connection) propagates unchanged for the caller's own handling."""
from agent.auxiliary_client import _drive_ladder
try:
return send(client, kwargs)
except Exception as first_err:
ladder = _parameter_ladder(first_err, client, kwargs, task=task, tag="")
return _drive_ladder(ladder, lambda step: send(*step.args))


async def send_with_parameter_rungs_async(
send: Callable[[Any, Dict[str, Any]], Awaitable[Any]], client: Any, kwargs: Dict[str, Any], *,
task: Optional[str],
) -> Any:
"""Async twin of :func:`send_with_parameter_rungs`."""
from agent.auxiliary_client import _drive_ladder_async
try:
return await send(client, kwargs)
except Exception as first_err:
ladder = _parameter_ladder(first_err, client, kwargs, task=task, tag=" (async)")
return await _drive_ladder_async(ladder, lambda step: send(*step.args))
7 changes: 6 additions & 1 deletion agent/title_generator.py
Original file line number Diff line number Diff line change
Expand Up @@ -310,11 +310,16 @@ def generate_title(
"__LANGUAGE_RULE__", _LANGUAGE_RULE_PINNED.format(language=language) if language else _LANGUAGE_RULE_MATCH_USER,
)
try:
# Use the provider's default temperature instead of forcing 0.3.
# Some models (e.g. GPT-5.6) only accept their server-side default
# and reject explicit temperature values, causing the daemon title
# thread to fail with "Unsupported value: 'temperature'".
# See: #72351, #51083, #51157
response = call_llm(
task="title_generation",
messages=[{"role": "system", "content": prompt}, {"role": "user", "content": user_snippet}],
# A title is a handful of tokens; a larger ceiling let chatty models burn seconds.
max_tokens=64, temperature=0.3, timeout=timeout, main_runtime=main_runtime,
max_tokens=64, temperature=None, timeout=timeout, main_runtime=main_runtime,
extra_body={"response_format": _TITLE_RESPONSE_FORMAT},
# The module contract above promises thinking-disabled operation,
# but nothing enforced it: with the aux default reasoning_effort
Expand Down
22 changes: 21 additions & 1 deletion plugins/model-providers/fireworks/__init__.py
Original file line number Diff line number Diff line change
@@ -1,12 +1,32 @@
"""Fireworks AI provider profile. Models are addressed by full catalog ID
(``accounts/fireworks/models/<slug>``), tracking fw-ai/fireconnect ``setup-cli``."""

from typing import Any

from hermes_cli import __version__ as _HERMES_VERSION
from providers import register_provider
from providers.base import ProviderProfile


fireworks = ProviderProfile(
class FireworksProfile(ProviderProfile):
"""Map Hermes reasoning controls onto Fireworks' OpenAI-compatible wire."""

def build_api_kwargs_extras(
self, *, reasoning_config: dict | None = None, **context: Any
) -> tuple[dict[str, Any], dict[str, Any]]:
# Fireworks rejects the nested ``extra_body.reasoning`` fallback some OpenAI-compatible
# gateways accept (#109774); its documented control is top-level ``reasoning_effort``
# (``none`` disables thinking). Overriding here marks the profile reasoning-aware, so the
# transport never sends the generic fallback on this route.
if not isinstance(reasoning_config, dict):
return {}, {}
if reasoning_config.get("enabled") is False:
return {}, {"reasoning_effort": "none"}
effort = reasoning_config.get("effort")
return {}, ({"reasoning_effort": effort} if effort else {})


fireworks = FireworksProfile(
name="fireworks", aliases=("fireworks-ai", "fw"), display_name="Fireworks AI",
description="Fireworks AI — OpenAI-compatible direct model API",
signup_url="https://app.fireworks.ai/settings/users/api-keys", env_vars=("FIREWORKS_API_KEY",),
Expand Down
Loading
Loading