From 03329c5832fe1db0b0ae8d22b15d6287244bbc83 Mon Sep 17 00:00:00 2001 From: JonthanaHanh <92574114+JonthanaHanh@users.noreply.github.com> Date: Mon, 27 Jul 2026 13:39:57 +0700 Subject: [PATCH 1/6] fix(agent): use provider-default temperature for title generation (#72351) Title generation hardcoded temperature=0.3 in its call_llm() call. Models like GPT-5.6 only accept their server-side default temperature and reject explicit values with "Unsupported value: 'temperature'". While call_llm() has a retry that strips temperature on error, the daemon thread races with session cleanup in short-lived CLI sessions, causing the retry to fail with a connection error. Fix: pass temperature=None so the provider uses its own default. This avoids the unsupported-temperature error entirely and eliminates the need for the retry path. Fixes #72351 --- agent/title_generator.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/agent/title_generator.py b/agent/title_generator.py index fdb0556e81510..b70f46a2d804a 100644 --- a/agent/title_generator.py +++ b/agent/title_generator.py @@ -310,11 +310,16 @@ def generate_title( "__LANGUAGE_RULE__", _LANGUAGE_RULE_PINNED.format(language=language) if language else _LANGUAGE_RULE_MATCH_USER, ) try: + # Use the provider's default temperature instead of forcing 0.3. + # Some models (e.g. GPT-5.6) only accept their server-side default + # and reject explicit temperature values, causing the daemon title + # thread to fail with "Unsupported value: 'temperature'". + # See: #72351, #51083, #51157 response = call_llm( task="title_generation", messages=[{"role": "system", "content": prompt}, {"role": "user", "content": user_snippet}], # A title is a handful of tokens; a larger ceiling let chatty models burn seconds. - max_tokens=64, temperature=0.3, timeout=timeout, main_runtime=main_runtime, + max_tokens=64, temperature=None, timeout=timeout, main_runtime=main_runtime, extra_body={"response_format": _TITLE_RESPONSE_FORMAT}, # The module contract above promises thinking-disabled operation, # but nothing enforced it: with the aux default reasoning_effort From 137284403fb23382426641de364aec1e651ee976 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Thu, 17 Sep 2026 00:50:37 -0700 Subject: [PATCH 2/6] fix: chain auxiliary parameter-rejection retries on primary and fallback requests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Reasoning models reject several request fields at once (gpt-5: temperature AND max_tokens, #78273), a reasoning-strip retry can then 400 on temperature (#72351), and strict-schema gateways such as Fireworks reject the generic extra_body.reasoning fallback with "Extra inputs are not permitted, field: 'reasoning'" (#109774) — a phrasing none of the unsupported-parameter predicates recognised, so the reasoning rung never fired and every title call 400'd. The parameter rungs were a fixed single-pass order (temperature → structured output → reasoning → max_tokens): a field rejected AFTER an earlier rung had already run was never stripped, and _param_rung_accepts did not admit a temperature 400 raised by a later retry. They now form a table walked until no rung matches, each field stripped at most once, so N rejected fields recover in N retries in whatever order the provider raises them. Fallback candidates (per-task chain, main chain, discovery) got a single shot with the caller's temperature/max_tokens/reasoning fields and raised on the first parameter 400; agent/auxiliary_fallback_recovery.py now runs the same parameter rungs around a candidate's request (sync + async), leaving auth / payment / connection errors to the caller's existing handling. Live: real gpt-5-mini title_generation call (reasoning_effort 400 → temperature 400 → 200), real gpt-5-mini fallback candidate sync+async (temperature 400 → 200), and a stand-in replaying Fireworks' documented 400 (reasoning stripped → 200) — all failed on origin/main. --- agent/auxiliary_client.py | 125 +++++++++++------- agent/auxiliary_fallback_recovery.py | 46 +++++++ .../test_auxiliary_parameter_rung_chaining.py | 79 +++++++++++ 3 files changed, 199 insertions(+), 51 deletions(-) create mode 100644 agent/auxiliary_fallback_recovery.py create mode 100644 tests/agent/test_auxiliary_parameter_rung_chaining.py diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index d4eda23737149..7816c3ad0b4ec 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -3187,6 +3187,8 @@ def _is_unsupported_parameter_error(exc: Exception, param: str) -> bool: "unsupported parameter", "unsupported_parameter", "not supported", "does not support", "doesn't support", "is deprecated for this model", "unknown parameter", "unrecognized request argument", "unrecognized parameter", "invalid parameter", + # Strict pydantic-validated gateways (Fireworks) name the unknown field this way (#109774). + "extra inputs are not permitted", )) @@ -3885,8 +3887,13 @@ def _send(client: Any, request_kwargs: Dict[str, Any], dest: _FallbackDestinatio ), task, ) + from agent.auxiliary_fallback_recovery import send_with_parameter_rungs + + def _send_recovering(client: Any, request_kwargs: Dict[str, Any], dest: _FallbackDestination) -> Any: + return send_with_parameter_rungs( + lambda c, kw: _send(c, kw, dest), client, request_kwargs, task=task) try: - return _send(fb_client, fb_kwargs, destination) + return _send_recovering(fb_client, fb_kwargs, destination) except Exception as fb_err: if not _is_auth_error(fb_err): raise @@ -3896,7 +3903,7 @@ def _send(client: Any, request_kwargs: Dict[str, Any], dest: _FallbackDestinatio if retry is not None: failed_destination = retry[2] try: - return _send(*retry) + return _send_recovering(*retry) except Exception as retry_err: if not _is_auth_error(retry_err): raise @@ -3924,8 +3931,13 @@ async def _send(client: Any, request_kwargs: Dict[str, Any], dest: _FallbackDest await _relay_async_completion(client, request_kwargs, provider=dest.provider, api_mode=dest.api_mode), task, ) + from agent.auxiliary_fallback_recovery import send_with_parameter_rungs_async + + async def _send_recovering(client: Any, request_kwargs: Dict[str, Any], dest: _FallbackDestination) -> Any: + return await send_with_parameter_rungs_async( + lambda c, kw: _send(c, kw, dest), client, request_kwargs, task=task) try: - return await _send(fb_client, fb_kwargs, destination) + return await _send_recovering(fb_client, fb_kwargs, destination) except Exception as fb_err: if not _is_auth_error(fb_err): raise @@ -3935,7 +3947,7 @@ async def _send(client: Any, request_kwargs: Dict[str, Any], dest: _FallbackDest if retry is not None: failed_destination = retry[2] try: - return await _send(*retry) + return await _send_recovering(*retry) except Exception as retry_err: if not _is_auth_error(retry_err): raise @@ -7010,8 +7022,10 @@ def _param_rung_accepts(exc: Exception) -> bool: chains with the stripped kwargs; re-raise anything those chains won't handle.""" return (_is_payment_error(exc) or _is_connection_error(exc) or _is_auth_error(exc) or "max_tokens" in str(exc) or "unsupported_parameter" in str(exc) - # Parameter rungs chain (temperature-strip retry 400s on reasoning_effort / response_format), - # and a route-gating 400 after a strip still reaches the provider-fallback rung. + # Parameter rungs chain in any order (a reasoning-strip retry can 400 on temperature, + # a temperature-strip retry on max_tokens), and a route-gating 400 after a strip still + # reaches the provider-fallback rung. + or _is_unsupported_parameter_error(exc, "temperature") or _is_reasoning_field_rejection(exc) or _is_structured_output_rejection(exc) or _is_model_incompatible_error(exc)) @@ -7030,61 +7044,70 @@ def _credential_rung_accepts(exc: Exception) -> bool: ]) +def _without_temperature(kwargs: dict) -> Optional[dict]: + """Copy *kwargs* without ``temperature``; None when it was not sent.""" + return {k: v for k, v in kwargs.items() if k != "temperature"} if "temperature" in kwargs else None + + +def _without_max_tokens(kwargs: dict) -> Optional[dict]: + """Copy *kwargs* without either output cap; None when neither was sent.""" + retry_kwargs = {k: v for k, v in kwargs.items() if k not in ("max_tokens", "max_completion_tokens")} + return retry_kwargs if len(retry_kwargs) != len(kwargs) else None + + +def _is_max_tokens_rejection(exc: Exception, client: Any) -> bool: + err_str = str(exc) + # ZAI vision models reject max_tokens with code 1210 and a message that never + # mentions "max_tokens", so detect it explicitly. + is_zai_param_error = "1210" in err_str and "bigmodel" in str(getattr(client, "base_url", "")) + return ("max_tokens" in err_str or "unsupported_parameter" in err_str + or _is_unsupported_parameter_error(exc, "max_tokens") or is_zai_param_error) + + +def _parameter_rungs(client: Any, max_tokens: Optional[int]) -> tuple: + """Ordered ``(matches, strip, log message)`` parameter rungs; ``strip`` returns None when the + field was not on the wire, so an unchanged request is never re-sent.""" + return ( + (lambda exc: _is_unsupported_parameter_error(exc, "temperature"), _without_temperature, + "provider rejected temperature; retrying without it"), + (_is_structured_output_rejection, _without_structured_output_format, + "provider rejected the structured-output format field; retrying without it " + "(schema enforcement degrades to prompt compliance)"), + # A chat-only model on an OpenAI-compatible relay rejects the profile's thinking-off encoding + # (top-level ``reasoning_effort: none``), and strict-schema gateways reject the generic + # ``extra_body.reasoning`` fallback outright (#109774); the caller only wanted "no thinking", + # so retry with every reasoning field omitted and let the route default apply (#112781). + (_is_reasoning_field_rejection, _without_reasoning_fields, + "provider rejected the reasoning field; retrying without it (route default applies)"), + (lambda exc: max_tokens is not None and _is_max_tokens_rejection(exc, client), _without_max_tokens, + "provider rejected the output cap; retrying without it"), + ) + + def _ladder_parameter_rungs( first_err: Exception, route: _LadderRoute, kwargs: Dict[str, Any], max_tokens: Optional[int], ): - """Rungs 1-4: retry without temperature / structured-output format / reasoning field / max_tokens. + """Parameter rungs: retry without temperature / structured-output format / reasoning field / + max_tokens. Rungs chain in whichever order the provider raises them (reasoning models reject + temperature AND max_tokens; a reasoning-strip retry can then trip temperature, #78273), each + field stripped at most once, so a request with N rejected fields recovers in N retries. Returns ``(response, None, kwargs)`` or ``(None, narrowed_err, stripped_kwargs)``.""" client, task, tag = route.client, route.task, route.tag - if "temperature" in kwargs and _is_unsupported_parameter_error(first_err, "temperature"): - retry_kwargs = {k: v for k, v in kwargs.items() if k != "temperature"} - logger.info("Auxiliary %s%s: provider rejected temperature; retrying once without it", - task or "call", tag) + rungs = list(_parameter_rungs(client, max_tokens)) + while rungs: + hit = next(((matches, strip, message) for matches, strip, message in rungs + if matches(first_err) and strip(kwargs) is not None), None) + if hit is None: + break + rungs.remove(hit) + matches, strip, message = hit + retry_kwargs = strip(kwargs) + logger.info("Auxiliary %s%s: %s: %s", task or "call", tag, message, first_err) resp, first_err = yield from _rung( _LadderStep("call", (client, retry_kwargs)), _param_rung_accepts) if first_err is None: return resp, None, retry_kwargs kwargs = retry_kwargs - if _is_structured_output_rejection(first_err): - retry_kwargs = _without_structured_output_format(kwargs) - if retry_kwargs is not None: - logger.info("Auxiliary %s%s: provider rejected the structured-output " - "format field; retrying once without it (schema " - "enforcement degrades to prompt compliance): %s", task or "call", tag, first_err) - resp, first_err = yield from _rung( - _LadderStep("call", (client, retry_kwargs)), _param_rung_accepts) - if first_err is None: - return resp, None, retry_kwargs - kwargs = retry_kwargs - # A chat-only model on an OpenAI-compatible relay rejects the profile's thinking-off encoding - # (top-level ``reasoning_effort: none``); the caller only wanted "no thinking", which is what - # such a model does anyway, so retry once with every reasoning field omitted (#112781). - if _is_reasoning_field_rejection(first_err): - retry_kwargs = _without_reasoning_fields(kwargs) - if retry_kwargs is not None: - logger.info("Auxiliary %s%s: provider rejected the reasoning field; retrying once " - "without it (route default applies): %s", task or "call", tag, first_err) - resp, first_err = yield from _rung( - _LadderStep("call", (client, retry_kwargs)), _param_rung_accepts) - if first_err is None: - return resp, None, retry_kwargs - kwargs = retry_kwargs - err_str = str(first_err) - # ZAI vision models reject max_tokens with code 1210 and a message that never - # mentions "max_tokens", so detect it explicitly. - _is_zai_param_error = "1210" in err_str and "bigmodel" in str(getattr(client, "base_url", "")) - if max_tokens is not None and ( - "max_tokens" in err_str or "unsupported_parameter" in err_str - or _is_unsupported_parameter_error(first_err, "max_tokens") or _is_zai_param_error - ): - kwargs.pop("max_tokens", None) - kwargs.pop("max_completion_tokens", None) - resp, first_err = yield from _rung( - _LadderStep("call", (client, kwargs)), - lambda exc: _is_payment_error(exc) or _is_connection_error(exc) or _is_rate_limit_error(exc), - ) - if first_err is None: - return resp, None, kwargs return None, first_err, kwargs diff --git a/agent/auxiliary_fallback_recovery.py b/agent/auxiliary_fallback_recovery.py new file mode 100644 index 0000000000000..12371d7d68a20 --- /dev/null +++ b/agent/auxiliary_fallback_recovery.py @@ -0,0 +1,46 @@ +"""Parameter-rejection recovery for auxiliary fallback candidates. + +The primary auxiliary request runs the full recovery ladder in ``auxiliary_client``; a fallback +candidate (per-task chain, main fallback chain, discovery) used to get a single shot, so a fallback +model that rejects ``temperature``, ``max_tokens`` or a reasoning field failed the whole task even +though the same parameter rungs would have recovered it on the primary path (#78273, #72351). +This module runs those rungs — and only those — around a candidate's request. +""" +from typing import Any, Awaitable, Callable, Dict, Optional + + +def _parameter_ladder(first_err: Exception, client: Any, kwargs: Dict[str, Any], *, + task: Optional[str], tag: str): + from agent.auxiliary_client import _LadderRoute, _ladder_parameter_rungs + route = _LadderRoute(client, task, tag, bool(tag), "", "", None, None, None, None, None, None, None) + max_tokens = kwargs.get("max_tokens") or kwargs.get("max_completion_tokens") + resp, err, _ = yield from _ladder_parameter_rungs(first_err, route, kwargs, max_tokens) + if err is None: + return resp + raise err + + +def send_with_parameter_rungs( + send: Callable[[Any, Dict[str, Any]], Any], client: Any, kwargs: Dict[str, Any], *, task: Optional[str], +) -> Any: + """``send(client, kwargs)``; on a parameter 400, retry through the parameter rungs. Any other + error (auth, payment, connection) propagates unchanged for the caller's own handling.""" + from agent.auxiliary_client import _drive_ladder + try: + return send(client, kwargs) + except Exception as first_err: + ladder = _parameter_ladder(first_err, client, kwargs, task=task, tag="") + return _drive_ladder(ladder, lambda step: send(*step.args)) + + +async def send_with_parameter_rungs_async( + send: Callable[[Any, Dict[str, Any]], Awaitable[Any]], client: Any, kwargs: Dict[str, Any], *, + task: Optional[str], +) -> Any: + """Async twin of :func:`send_with_parameter_rungs`.""" + from agent.auxiliary_client import _drive_ladder_async + try: + return await send(client, kwargs) + except Exception as first_err: + ladder = _parameter_ladder(first_err, client, kwargs, task=task, tag=" (async)") + return await _drive_ladder_async(ladder, lambda step: send(*step.args)) diff --git a/tests/agent/test_auxiliary_parameter_rung_chaining.py b/tests/agent/test_auxiliary_parameter_rung_chaining.py new file mode 100644 index 0000000000000..7ebae5d0642fd --- /dev/null +++ b/tests/agent/test_auxiliary_parameter_rung_chaining.py @@ -0,0 +1,79 @@ +"""Auxiliary parameter-rejection rungs chain in any order, on the primary AND the fallback path. + +Reasoning models reject several request fields at once (``temperature`` and ``max_tokens`` on +gpt-5, #78273), a reasoning-strip retry can then 400 on ``temperature`` (#72351), and strict-schema +gateways phrase an unknown ``reasoning`` field as "Extra inputs are not permitted" (#109774). +Each rejected field must be stripped in whatever order the provider raises them, and a fallback +candidate must get the same recovery as the primary request. +""" +from types import SimpleNamespace +from unittest.mock import MagicMock + +from agent.auxiliary_client import ( + _call_fallback_candidate_sync, + _drive_ladder, + _is_reasoning_field_rejection, + _ladder_parameter_rungs, + _LadderRoute, +) + + +class _Bad400(Exception): + status_code = 400 + + +def _ok(text="ok"): + return SimpleNamespace(choices=[SimpleNamespace(message=SimpleNamespace(content=text), finish_reason="stop")]) + + +def _rejecting_client(*rejected_fields): + """Fake OpenAI client that 400s while any of ``rejected_fields`` is still on the wire.""" + client = MagicMock(base_url="https://api.example/v1") + + def create(**kwargs): + body = dict(kwargs) + body.update(body.pop("extra_body", None) or {}) + for field in rejected_fields: + if field in body: + phrase = { + "reasoning": "Extra inputs are not permitted, field: 'reasoning'", + "reasoning_effort": "Unsupported value: 'reasoning_effort' does not support 'none' with this model.", + "temperature": "Unsupported value: 'temperature' does not support 0.3 with this model.", + "max_tokens": "Unsupported parameter: 'max_tokens' is not supported with this model.", + }[field] + raise _Bad400(f"Error code: 400 - {phrase}") + return _ok() + + client.chat.completions.create.side_effect = create + return client + + +def test_primary_rungs_chain_in_provider_order_and_strip_each_field_once(): + client = _rejecting_client("reasoning_effort", "temperature", "max_tokens") + kwargs = {"model": "m", "messages": [], "temperature": 0.3, "max_tokens": 64, "reasoning_effort": "none"} + route = _LadderRoute(client, "title_generation", "", False, "", "", None, None, None, None, None, None, None) + first_err = _Bad400("Error code: 400 - Unsupported value: 'reasoning_effort' does not support 'none'") + + def perform(step): + step_client, step_kwargs = step.args + return step_client.chat.completions.create(**step_kwargs) + + resp, err, final_kwargs = _drive_ladder(_ladder_parameter_rungs(first_err, route, kwargs, 64), perform) + + assert err is None and resp.choices[0].message.content == "ok" + assert not {"temperature", "max_tokens", "reasoning_effort"} & set(final_kwargs) + # One retry per rejected field: reasoning_effort → temperature → max_tokens, nothing re-sent unchanged. + assert client.chat.completions.create.call_count == 3 + assert _is_reasoning_field_rejection(_Bad400("Extra inputs are not permitted, field: 'reasoning'")) + + +def test_fallback_candidate_recovers_from_rejected_temperature(): + client = _rejecting_client("temperature") + resp = _call_fallback_candidate_sync( + client, "gpt-5-mini", "fallback_chain[0](openai)", task="title_generation", + messages=[{"role": "user", "content": "hi"}], temperature=0.3, max_tokens=16, tools=None, + effective_timeout=30.0, effective_extra_body={}, reasoning_config=None, + ) + assert resp.choices[0].message.content == "ok" + sent = [c.kwargs for c in client.chat.completions.create.call_args_list] + assert [("temperature" in k) for k in sent] == [True, False] From 58fe4efe5b4c57a87aef5009adece7e46648ab24 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Thu, 17 Sep 2026 01:15:54 -0700 Subject: [PATCH 3/6] fix(aux): build the fallback ladder route by field name A positional _LadderRoute(...) in auxiliary_fallback_recovery breaks as soon as the route tuple gains a field (#113968 adds timeout); construct by name so fields the parameter ladder never reads default to None regardless of tuple width. --- agent/auxiliary_fallback_recovery.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/agent/auxiliary_fallback_recovery.py b/agent/auxiliary_fallback_recovery.py index 12371d7d68a20..cce6e93363320 100644 --- a/agent/auxiliary_fallback_recovery.py +++ b/agent/auxiliary_fallback_recovery.py @@ -12,7 +12,10 @@ def _parameter_ladder(first_err: Exception, client: Any, kwargs: Dict[str, Any], *, task: Optional[str], tag: str): from agent.auxiliary_client import _LadderRoute, _ladder_parameter_rungs - route = _LadderRoute(client, task, tag, bool(tag), "", "", None, None, None, None, None, None, None) + # Keyword construction: the route tuple grows with every new ladder rung (a positional 13-tuple + # broke the moment a sibling PR added ``timeout``); fields this ladder never reads stay None. + route = _LadderRoute(**{**dict.fromkeys(_LadderRoute._fields), "client": client, "task": task, + "tag": tag, "async_mode": bool(tag), "base_info": "", "resolved_provider": ""}) max_tokens = kwargs.get("max_tokens") or kwargs.get("max_completion_tokens") resp, err, _ = yield from _ladder_parameter_rungs(first_err, route, kwargs, max_tokens) if err is None: From c187a45528edf388888dd2fc4b1d1fe98696a01a Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Thu, 17 Sep 2026 01:18:24 -0700 Subject: [PATCH 4/6] fix(aux): fallback ladder route carries the client base_url; test builds the route by name base_info is the endpoint identity every rung reads for per-route bookkeeping; the fallback ladder left it empty. The test mirrored the positional construction that broke under a wider tuple. --- agent/auxiliary_fallback_recovery.py | 3 ++- tests/agent/test_auxiliary_parameter_rung_chaining.py | 3 ++- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/agent/auxiliary_fallback_recovery.py b/agent/auxiliary_fallback_recovery.py index cce6e93363320..5875348a7cbca 100644 --- a/agent/auxiliary_fallback_recovery.py +++ b/agent/auxiliary_fallback_recovery.py @@ -15,7 +15,8 @@ def _parameter_ladder(first_err: Exception, client: Any, kwargs: Dict[str, Any], # Keyword construction: the route tuple grows with every new ladder rung (a positional 13-tuple # broke the moment a sibling PR added ``timeout``); fields this ladder never reads stay None. route = _LadderRoute(**{**dict.fromkeys(_LadderRoute._fields), "client": client, "task": task, - "tag": tag, "async_mode": bool(tag), "base_info": "", "resolved_provider": ""}) + "tag": tag, "async_mode": bool(tag), "resolved_provider": "", + "base_info": str(getattr(client, "base_url", "") or "")}) max_tokens = kwargs.get("max_tokens") or kwargs.get("max_completion_tokens") resp, err, _ = yield from _ladder_parameter_rungs(first_err, route, kwargs, max_tokens) if err is None: diff --git a/tests/agent/test_auxiliary_parameter_rung_chaining.py b/tests/agent/test_auxiliary_parameter_rung_chaining.py index 7ebae5d0642fd..ee7dc44ef9339 100644 --- a/tests/agent/test_auxiliary_parameter_rung_chaining.py +++ b/tests/agent/test_auxiliary_parameter_rung_chaining.py @@ -51,7 +51,8 @@ def create(**kwargs): def test_primary_rungs_chain_in_provider_order_and_strip_each_field_once(): client = _rejecting_client("reasoning_effort", "temperature", "max_tokens") kwargs = {"model": "m", "messages": [], "temperature": 0.3, "max_tokens": 64, "reasoning_effort": "none"} - route = _LadderRoute(client, "title_generation", "", False, "", "", None, None, None, None, None, None, None) + route = _LadderRoute(**{**dict.fromkeys(_LadderRoute._fields), "client": client, "task": "title_generation", + "tag": "", "async_mode": False, "base_info": "", "resolved_provider": ""}) first_err = _Bad400("Error code: 400 - Unsupported value: 'reasoning_effort' does not support 'none'") def perform(step): From 99ff3b921db31f32e21765b42656d9101efd1a80 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Thu, 17 Sep 2026 01:35:13 -0700 Subject: [PATCH 5/6] fix(aux): parameter rungs hand a 429 on, and Fireworks gets top-level reasoning_effort The pre-ladder max_tokens rung accepted payment|connection|rate_limit errors on its retry, so a 429 after a stripped retry reached the credential and provider-fallback rungs. The ladder's shared `_param_rung_accepts` dropped rate_limit: a 429 on the retry raised out of the primary call and skipped the whole fallback chain. Accept it there too, with a rung-level test that the stripped kwargs and the 429 are handed on rather than raised. The body claimed `Fixes #109774` while the aux client still sent the generic `extra_body.reasoning` to Fireworks on every call and only recovered reactively (an extra 400 round-trip per request). Port @huklaa's profile override from #109807: Fireworks documents top-level `reasoning_effort` (`none` disables thinking), and overriding `build_api_kwargs_extras` marks the profile reasoning-aware so the transport omits the generic fallback on this route. Extends the salvage with the effort mapping so enabled-with-effort takes the same wire, with a control that a profile-less route keeps the fallback. Salvages #109807 (@huklaa). Co-authored-by: Hukla <129692708+huklaa@users.noreply.github.com> --- agent/auxiliary_client.py | 5 +++- plugins/model-providers/fireworks/__init__.py | 22 ++++++++++++++- .../test_auxiliary_parameter_rung_chaining.py | 25 +++++++++++++++++ .../model_providers/test_fireworks_profile.py | 28 +++++++++++++++++++ 4 files changed, 78 insertions(+), 2 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 7816c3ad0b4ec..a7550e2a5febf 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -7019,8 +7019,11 @@ def _rung(step: "_LadderStep", accept: Callable[[Exception], bool]): def _param_rung_accepts(exc: Exception) -> bool: """After a parameter-strip retry: fall through to the max_tokens/payment/auth - chains with the stripped kwargs; re-raise anything those chains won't handle.""" + chains with the stripped kwargs; re-raise anything those chains won't handle. + A 429 on the retry is the credential/provider-fallback rungs' job, so it falls + through too (the pre-ladder max_tokens rung accepted rate limits).""" return (_is_payment_error(exc) or _is_connection_error(exc) or _is_auth_error(exc) + or _is_rate_limit_error(exc) or "max_tokens" in str(exc) or "unsupported_parameter" in str(exc) # Parameter rungs chain in any order (a reasoning-strip retry can 400 on temperature, # a temperature-strip retry on max_tokens), and a route-gating 400 after a strip still diff --git a/plugins/model-providers/fireworks/__init__.py b/plugins/model-providers/fireworks/__init__.py index 3d57f99b63017..4c0496bbd1829 100644 --- a/plugins/model-providers/fireworks/__init__.py +++ b/plugins/model-providers/fireworks/__init__.py @@ -1,12 +1,32 @@ """Fireworks AI provider profile. Models are addressed by full catalog ID (``accounts/fireworks/models/``), tracking fw-ai/fireconnect ``setup-cli``.""" +from typing import Any + from hermes_cli import __version__ as _HERMES_VERSION from providers import register_provider from providers.base import ProviderProfile -fireworks = ProviderProfile( +class FireworksProfile(ProviderProfile): + """Map Hermes reasoning controls onto Fireworks' OpenAI-compatible wire.""" + + def build_api_kwargs_extras( + self, *, reasoning_config: dict | None = None, **context: Any + ) -> tuple[dict[str, Any], dict[str, Any]]: + # Fireworks rejects the nested ``extra_body.reasoning`` fallback some OpenAI-compatible + # gateways accept (#109774); its documented control is top-level ``reasoning_effort`` + # (``none`` disables thinking). Overriding here marks the profile reasoning-aware, so the + # transport never sends the generic fallback on this route. + if not isinstance(reasoning_config, dict): + return {}, {} + if reasoning_config.get("enabled") is False: + return {}, {"reasoning_effort": "none"} + effort = reasoning_config.get("effort") + return {}, ({"reasoning_effort": effort} if effort else {}) + + +fireworks = FireworksProfile( name="fireworks", aliases=("fireworks-ai", "fw"), display_name="Fireworks AI", description="Fireworks AI — OpenAI-compatible direct model API", signup_url="https://app.fireworks.ai/settings/users/api-keys", env_vars=("FIREWORKS_API_KEY",), diff --git a/tests/agent/test_auxiliary_parameter_rung_chaining.py b/tests/agent/test_auxiliary_parameter_rung_chaining.py index ee7dc44ef9339..dd77fd1fa6ef4 100644 --- a/tests/agent/test_auxiliary_parameter_rung_chaining.py +++ b/tests/agent/test_auxiliary_parameter_rung_chaining.py @@ -78,3 +78,28 @@ def test_fallback_candidate_recovers_from_rejected_temperature(): assert resp.choices[0].message.content == "ok" sent = [c.kwargs for c in client.chat.completions.create.call_args_list] assert [("temperature" in k) for k in sent] == [True, False] + + +def test_rate_limit_after_parameter_strip_falls_through_to_later_rungs(): + """A 429 on the stripped retry belongs to the credential/provider-fallback rungs; the + parameter rungs must hand it on with the stripped kwargs, not raise out of the ladder.""" + import httpx + import openai + + request = httpx.Request("POST", "https://api.example/v1/chat/completions") + rate_limited = openai.RateLimitError( + "Error code: 429 - Rate limit exceeded", body=None, + response=httpx.Response(429, request=request, json={"error": {"message": "Rate limit exceeded"}})) + client = MagicMock(base_url="https://api.example/v1") + client.chat.completions.create.side_effect = rate_limited + route = _LadderRoute(**{**dict.fromkeys(_LadderRoute._fields), "client": client, "task": "title_generation", + "tag": "", "async_mode": False, "base_info": "", "resolved_provider": ""}) + kwargs = {"model": "m", "messages": [], "max_tokens": 64} + first_err = _Bad400("Error code: 400 - Unsupported parameter: 'max_tokens' is not supported with this model.") + + resp, err, final_kwargs = _drive_ladder( + _ladder_parameter_rungs(first_err, route, kwargs, 64), + lambda step: step.args[0].chat.completions.create(**step.args[1])) + + assert resp is None and err is rate_limited + assert "max_tokens" not in final_kwargs diff --git a/tests/plugins/model_providers/test_fireworks_profile.py b/tests/plugins/model_providers/test_fireworks_profile.py index bf12649adce36..cb12cbff86ee8 100644 --- a/tests/plugins/model_providers/test_fireworks_profile.py +++ b/tests/plugins/model_providers/test_fireworks_profile.py @@ -92,3 +92,31 @@ def test_fallback_models_are_payg_models_not_routers(self, fireworks_profile): assert model.startswith("accounts/fireworks/models/"), model assert "/routers/" not in model assert "turbo" not in model.lower(), model + + +class TestFireworksReasoning: + @pytest.mark.parametrize( + "provider, reasoning_config, expect_top_level, expect_generic", + [ + # Fireworks: thinking-off goes out as the documented top-level control, never as the + # nested ``extra_body.reasoning`` the API 400s on (#109774, salvaged from #109807). + ("fireworks", {"enabled": False}, {"reasoning_effort": "none"}, False), + ("fireworks", {"enabled": True, "effort": "low"}, {"reasoning_effort": "low"}, False), + # Control: a route without a reasoning-aware profile keeps the generic fallback. + ("unregistered-gateway", {"enabled": False}, {}, True), + ], + ) + def test_auxiliary_reasoning_wire_shape( + self, fireworks_profile, provider, reasoning_config, expect_top_level, expect_generic + ): + from agent.auxiliary_client import _build_call_kwargs + + kwargs = _build_call_kwargs( + provider, "accounts/fireworks/models/glm-5p2", + [{"role": "user", "content": "Generate a title"}], + reasoning_config=reasoning_config, base_url="https://api.fireworks.ai/inference/v1", + task="title_generation", + ) + + assert {k: v for k, v in kwargs.items() if k == "reasoning_effort"} == expect_top_level + assert ("reasoning" in kwargs.get("extra_body", {})) is expect_generic From 9e6a818e28fa2c755f43be674d0172a2aa4fbd97 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Thu, 17 Sep 2026 01:51:19 -0700 Subject: [PATCH 6/6] fix(aux): max_tokens rung retries even when the wire kwargs no longer carry the cap The rung table refused to re-send an unchanged request, but the Codex Responses route translates the caller cap away and gateways inject their own: the 400 names max_tokens while the kwargs show none, and the identical retry is what completes (tests/agent/test_injected_param_strip_retry_registry.py, red in CI on 99ff3b9). --- agent/auxiliary_client.py | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index a7550e2a5febf..8c09d193fda50 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -7052,10 +7052,12 @@ def _without_temperature(kwargs: dict) -> Optional[dict]: return {k: v for k, v in kwargs.items() if k != "temperature"} if "temperature" in kwargs else None -def _without_max_tokens(kwargs: dict) -> Optional[dict]: - """Copy *kwargs* without either output cap; None when neither was sent.""" - retry_kwargs = {k: v for k, v in kwargs.items() if k not in ("max_tokens", "max_completion_tokens")} - return retry_kwargs if len(retry_kwargs) != len(kwargs) else None +def _without_max_tokens(kwargs: dict) -> dict: + """Copy *kwargs* without either output cap. Unlike the other strips this never returns None: a + route can translate the caller's cap into a field the wire kwargs no longer show (Codex + Responses), and the provider's own gateway may inject one — the 400 still names ``max_tokens`` + and the identical request completes on retry (registry class #89897/#90257).""" + return {k: v for k, v in kwargs.items() if k not in ("max_tokens", "max_completion_tokens")} def _is_max_tokens_rejection(exc: Exception, client: Any) -> bool: