From f7fc4e3b64d518bb4cbd63c03bd1004e539c016a Mon Sep 17 00:00:00 2001 From: buruguo Date: Wed, 15 Apr 2026 09:14:25 +0800 Subject: [PATCH] [Bug] Qwen3 reasoning detector swallows tool_call when is missing Qwen3Detector forwards every constructor argument to the base class except tool_start_token. The base BaseReasoningFormatDetector already contains a fallback that, when in reasoning state and a tool start token appears, forwards the tool call to normal_text instead of keeping it inside reasoning_content. Because Qwen3Detector never sets this token, the fallback is silently disabled for every Qwen3 family model that uses the qwen3 reasoning parser. In production this is reproducible with Qwen3.5-27B served via SGLang: 1. force_reasoning=True (e.g. enable_thinking=True request, the default for Qwen3-style chat templates that auto-prepend ) 2. The model emits ... directly without first emitting (long contexts, certain tool-call code templates, or when the model decides not to think before acting) 3. The entire response is silently routed into reasoning_content, content is null, and tool_calls is null 4. Downstream the function-call parser sees no tool calls and LangChain-style frameworks raise a Pydantic validation error ("ToolMessage tool_call_id None") because the framework tries to construct a ToolMessage from the empty tool_call Fix: pass tool_start_token="" to the base class so the existing fallback is wired up. The fallback only triggers when _in_reasoning=True, so behavior is unchanged for enable_thinking=False requests and for normal flows that include a proper . Added two regression tests under TestQwen3ForcedReasoningDetector covering both the non-streaming and streaming paths. --- python/sglang/srt/parser/reasoning_parser.py | 1 + .../unit/parser/test_reasoning_parser.py | 42 +++++++++++++++++++ 2 files changed, 43 insertions(+) diff --git a/python/sglang/srt/parser/reasoning_parser.py b/python/sglang/srt/parser/reasoning_parser.py index 8811c90b2ddc..b55e37a383cf 100644 --- a/python/sglang/srt/parser/reasoning_parser.py +++ b/python/sglang/srt/parser/reasoning_parser.py @@ -242,6 +242,7 @@ def __init__( "", force_reasoning=force_reasoning, stream_reasoning=stream_reasoning, + tool_start_token="", continue_final_message=continue_final_message, previous_content=previous_content, ) diff --git a/test/registered/unit/parser/test_reasoning_parser.py b/test/registered/unit/parser/test_reasoning_parser.py index 92ad536a872c..8a06a81074e8 100644 --- a/test/registered/unit/parser/test_reasoning_parser.py +++ b/test/registered/unit/parser/test_reasoning_parser.py @@ -269,6 +269,48 @@ def test_streaming_qwen3_forced_reasoning_format(self): self.assertEqual(result.reasoning_text, "") # Buffer cleared self.assertEqual(result.normal_text, "The answer is 42.") + def test_detect_and_parse_tool_call_without_think_close(self): + """ + Regression test: when force_reasoning=True and the model emits + without first closing , the tool_call must be split into normal_text + so the downstream tool-call parser can still see it. Otherwise the entire + output is silently swallowed into reasoning_content and the function call + is lost (observed with Qwen3.5-27B serving via SGLang in production). + """ + text = "I should call the tool.\n\n\n" + result = self.detector.detect_and_parse(text) + self.assertEqual(result.reasoning_text, "I should call the tool.") + self.assertEqual( + result.normal_text, + "\n\n\n", + ) + + def test_streaming_tool_call_without_think_close(self): + """ + Streaming regression: same scenario as above but for incremental parsing. + Once appears while still in reasoning state, the parser must + flip to normal_text and forward ... downstream. + """ + # Initial reasoning chunks (no ) + result = self.detector.parse_streaming_increment("Let me ") + self.assertEqual(result.reasoning_text, "Let me ") + self.assertEqual(result.normal_text, "") + + result = self.detector.parse_streaming_increment("call the tool.") + self.assertEqual(result.reasoning_text, "call the tool.") + self.assertEqual(result.normal_text, "") + + # Tool call appears WITHOUT a preceding + result = self.detector.parse_streaming_increment( + "\n\n\n" + ) + self.assertEqual(result.reasoning_text, "") + self.assertEqual( + result.normal_text, + "\n\n\n", + ) + self.assertFalse(self.detector._in_reasoning) + class TestKimiDetector(CustomTestCase): def setUp(self):