From 8057c87eec8e73a6b680e26e8b94bc82f3345f7f Mon Sep 17 00:00:00 2001 From: vominh1919 Date: Wed, 6 May 2026 13:50:04 +0700 Subject: [PATCH] fix: forward reasoning_config to custom providers (vLLM, Ollama, etc.) Fixes #20576 When using provider: custom with a vLLM-served thinking model (e.g. MiniMax-M2.7, DeepSeek-R1, GLM-4.x), the agent silently sent no thinking-budget control on the wire request because: 1. _supports_reasoning_extra_body() returned False for any base_url not in its hardcoded allowlist (OpenRouter, Nous, GitHub, LM Studio), so extra_body.reasoning was never emitted. 2. Even when reasoning was supported, the effort level was hardcoded to 'medium' instead of respecting the user's reasoning_config.effort. Root cause: _supports_reasoning_extra_body() had no check for custom providers, silently dropping the user's explicit reasoning_config. Fix: - In run_agent.py: add a 'custom' provider check that honors the user's reasoning_config (returns True when enabled is not explicitly False). - In chat_completions.py: use the user's configured effort level from reasoning_config instead of hardcoding 'medium'. This allows vLLM's reasoning_effort and thinking_token_budget parameters to be forwarded correctly, preventing runaway reasoning that consumes the entire max_tokens budget inside blocks. --- agent/transports/chat_completions.py | 11 ++++++++++- run_agent.py | 8 ++++++++ 2 files changed, 18 insertions(+), 1 deletion(-) diff --git a/agent/transports/chat_completions.py b/agent/transports/chat_completions.py index ca29b39ffe4e6..f6dc54e407df7 100644 --- a/agent/transports/chat_completions.py +++ b/agent/transports/chat_completions.py @@ -341,7 +341,16 @@ def build_kwargs( if gh_reasoning is not None: extra_body["reasoning"] = gh_reasoning else: - extra_body["reasoning"] = {"enabled": True, "effort": "medium"} + # Use the user-configured effort level when available, + # falling back to "medium" for known providers. Custom + # providers (vLLM etc.) carry the user's reasoning_config + # effort so thinking_token_budget is respected. See #20576. + _effort = "medium" + if reasoning_config and isinstance(reasoning_config, dict): + _e = (reasoning_config.get("effort") or "").strip().lower() + if _e in ("low", "medium", "high"): + _effort = _e + extra_body["reasoning"] = {"enabled": True, "effort": _effort} if provider_name == "gemini": raw_thinking_config = _build_gemini_thinking_config(model, reasoning_config) diff --git a/run_agent.py b/run_agent.py index 0b69a17175257..a72cdd46b14c1 100644 --- a/run_agent.py +++ b/run_agent.py @@ -8662,6 +8662,14 @@ def _supports_reasoning_extra_body(self) -> bool: opts = self._lmstudio_reasoning_options_cached() # "off-only" (or absent) means no real reasoning capability. return any(opt and opt != "off" for opt in opts) + # Custom provider (e.g. vLLM, Ollama, etc.): honor explicit + # reasoning_config from the user. vLLM supports reasoning_effort + # and thinking_token_budget for thinking models; without this gate + # those parameters are silently dropped. See #20576. + if (self.provider or "").strip().lower() == "custom": + if self.reasoning_config and isinstance(self.reasoning_config, dict): + return self.reasoning_config.get("enabled") is not False + return False if "openrouter" not in self._base_url_lower: return False if "api.mistral.ai" in self._base_url_lower: