From 423d6378232eb82384bff9c35174df21122e1cee Mon Sep 17 00:00:00 2001 From: wjameswen888 Date: Fri, 8 May 2026 12:53:26 +0900 Subject: [PATCH] fix(deepseek): send explicit thinking:disabled when reasoning is off MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit DeepSeek direct API requires an explicit thinking.type parameter. When thinking is disabled (reasoning_effort=none or enabled=false), the API defaults to thinking=enabled — generating reasoning_content that gets stripped/not passed back, causing HTTP 400 on subsequent turns. This mirrors the existing Kimi extra_body.thinking pattern: - Detect DeepSeek provider/model/base_url in run_agent.py - Send {"thinking": {"type": "disabled"}} when appropriate - Respect reasoning_config.effort=none as disabled Fixes: #15700, #17212 Related: #15213 --- agent/transports/chat_completions.py | 14 ++++++++++++++ run_agent.py | 6 ++++++ sessions.db | 0 3 files changed, 20 insertions(+) create mode 100644 sessions.db diff --git a/agent/transports/chat_completions.py b/agent/transports/chat_completions.py index ca29b39ffe4e..1c21502679ca 100644 --- a/agent/transports/chat_completions.py +++ b/agent/transports/chat_completions.py @@ -258,6 +258,7 @@ def build_kwargs( anthropic_max_out = params.get("anthropic_max_output") is_nvidia_nim = params.get("is_nvidia_nim", False) is_kimi = params.get("is_kimi", False) + is_deepseek = params.get("is_deepseek", False) is_tokenhub = params.get("is_tokenhub", False) reasoning_config = params.get("reasoning_config") @@ -333,6 +334,19 @@ def build_kwargs( "type": "enabled" if _kimi_thinking_enabled else "disabled", } + # DeepSeek extra_body.thinking (same format as Kimi) + if is_deepseek: + _ds_thinking_enabled = True + if reasoning_config and isinstance(reasoning_config, dict): + if reasoning_config.get("enabled") is False: + _ds_thinking_enabled = False + _ds_effort = (reasoning_config.get("effort") or "").strip().lower() + if _ds_effort == "none": + _ds_thinking_enabled = False + extra_body["thinking"] = { + "type": "enabled" if _ds_thinking_enabled else "disabled", + } + # Reasoning. LM Studio is handled above via top-level reasoning_effort, # so skip emitting extra_body.reasoning for it. if params.get("supports_reasoning", False) and not params.get("is_lmstudio", False): diff --git a/run_agent.py b/run_agent.py index 403dba4e7850..f00621ee14c2 100644 --- a/run_agent.py +++ b/run_agent.py @@ -8710,6 +8710,11 @@ def _build_api_kwargs(self, api_messages: list) -> dict: or base_url_host_matches(self.base_url, "moonshot.cn") ) _is_tokenhub = base_url_host_matches(self._base_url_lower, "tokenhub.tencentmaas.com") + _is_deepseek = ( + self.provider == "deepseek" + or "deepseek" in (self.model or "").lower() + or base_url_host_matches(self.base_url, "api.deepseek.com") + ) _is_lmstudio = (self.provider or "").strip().lower() == "lmstudio" # Temperature: _fixed_temperature_for_model may return OMIT_TEMPERATURE @@ -8819,6 +8824,7 @@ def _build_api_kwargs(self, api_messages: list) -> dict: is_github_models=_is_gh, is_nvidia_nim=_is_nvidia, is_kimi=_is_kimi, + is_deepseek=_is_deepseek, is_tokenhub=_is_tokenhub, is_lmstudio=_is_lmstudio, is_custom_provider=self.provider == "custom", diff --git a/sessions.db b/sessions.db new file mode 100644 index 000000000000..e69de29bb2d1