diff --git a/tests/parser/engine/test_deepseek_v32.py b/tests/parser/engine/test_deepseek_v32.py index 7825f5add836..55ceb0ba4162 100644 --- a/tests/parser/engine/test_deepseek_v32.py +++ b/tests/parser/engine/test_deepseek_v32.py @@ -22,6 +22,7 @@ DSML_INVOKE_END, DSML_INVOKE_NAME_END, DSML_INVOKE_PREFIX, + DSML_TOOL_START, ) from vllm.parser.deepseek_v32 import ( DSML_FUNC_END, @@ -67,13 +68,7 @@ def _make_tool(name, properties): ) -@pytest.fixture -def mock_tokenizer(): - return make_mock_tokenizer({}) - - -@pytest.fixture -def mock_request(): +def _request_without_tools(): from unittest.mock import MagicMock from vllm.entrypoints.openai.chat_completion.protocol import ( @@ -86,6 +81,16 @@ def mock_request(): return req +@pytest.fixture +def mock_tokenizer(): + return make_mock_tokenizer({}) + + +@pytest.fixture +def mock_request(): + return _request_without_tools() + + # ── Non-streaming extraction ──────────────────────────────────────── @@ -130,6 +135,67 @@ def test_content_before_tool_call(self, mock_tokenizer, mock_request): assert result.content is not None assert "Let me check" in result.content + def test_missing_func_start_orphan_invoke(self, mock_tokenizer, mock_request): + """Orphan invoke without the <|DSML|function_calls> wrapper is + still parsed as a tool call when the request declared the tool + (see gh-48931).""" + tool = _make_tool("get_weather", {"city": {"type": "string"}}) + mock_request.tools = [tool] + text = _invoke("get_weather", _param("city", "true", "SF")) + DSML_FUNC_END + parser = DeepSeekV32Parser(mock_tokenizer, tools=[tool]) + result = parser.extract_tool_calls(text, mock_request) + assert result.tools_called + assert len(result.tool_calls) == 1 + assert result.tool_calls[0].function.name == "get_weather" + args = json.loads(result.tool_calls[0].function.arguments) + assert args == {"city": "SF"} + assert result.content is None + + def test_orphan_invoke_without_declared_tools_stays_content( + self, mock_tokenizer, mock_request + ): + """A request that declared no tools can never accept a recovered + name, so the orphan invoke stays plain content.""" + text = _invoke("get_weather", _param("city", "true", "SF")) + DSML_FUNC_END + parser = DeepSeekV32Parser(mock_tokenizer) + result = parser.extract_tool_calls(text, mock_request) + assert not result.tools_called + assert result.tool_calls == [] + assert result.content == text + + def test_unclosed_foreign_wrapper_then_native_call( + self, mock_tokenizer, mock_request + ): + """A foreign wrapper that never closes must not disable native + tool parsing: the token backed function_calls wrapper still + wins.""" + text = ( + DSML_TOOL_START + + "\nStray foreign text.\n" + + _func_calls(_invoke("get_weather", _param("city", "true", "SF"))) + ) + parser = DeepSeekV32Parser(mock_tokenizer) + result = parser.extract_tool_calls(text, mock_request) + assert result.tools_called + assert len(result.tool_calls) == 1 + assert result.tool_calls[0].function.name == "get_weather" + args = json.loads(result.tool_calls[0].function.arguments) + assert args == {"city": "SF"} + assert "Stray foreign text." in result.content + + def test_foreign_tool_calls_wrapper_rejected(self, mock_tokenizer, mock_request): + """An invoke inside the V4-style tool_calls wrapper stays plain + content: the orphan fallback must not fire inside a foreign + wrapper.""" + text = _func_calls( + _invoke("get_weather", _param("city", "true", "SF")), + ).replace("function_calls", "tool_calls") + parser = DeepSeekV32Parser(mock_tokenizer) + result = parser.extract_tool_calls(text, mock_request) + assert not result.tools_called + assert result.tool_calls == [] + assert result.content == text + def test_non_string_params_json_parsed(self, mock_tokenizer, mock_request): text = _func_calls( _invoke( @@ -159,6 +225,418 @@ def test_wrapper_unwrapping(self, mock_tokenizer, mock_request): assert args == {"location": "Beijing"} +# ── Orphan invoke name validation ──────────────────────────────────── + + +class TestOrphanInvokeNameValidation: + """Recovered (orphan) invokes must carry a plausible tool name. + + Mirrors the V4 coverage: when the request declares tools, the + (CONTENT, INVOKE_PREFIX) recovery path only commits to a tool call + if the parsed name is one of the declared functions; otherwise the + consumed text is re-emitted as plain content. + """ + + @pytest.fixture + def weather_tool(self): + return _make_tool("get_weather", {"city": {"type": "string"}}) + + def test_declared_name_recovered(self, mock_tokenizer, mock_request, weather_tool): + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool]) + mock_request.tools = [weather_tool] + text = _invoke("get_weather", _param("city", "true", "SF")) + DSML_FUNC_END + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called + assert len(result.tool_calls) == 1 + assert result.tool_calls[0].function.name == "get_weather" + args = json.loads(result.tool_calls[0].function.arguments) + assert args == {"city": "SF"} + assert result.content is None + + def test_undeclared_name_stays_content( + self, mock_tokenizer, mock_request, weather_tool + ): + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool]) + mock_request.tools = [weather_tool] + text = ( + "Quoting " + + DSML_INVOKE_PREFIX + + "made_up_tool" + + DSML_INVOKE_NAME_END + + " literally." + ) + result = parser.extract_tool_calls(text, mock_request) + + assert not result.tools_called + assert result.tool_calls == [] + assert result.content == text + + def test_char_by_char_undeclared_name_stays_content( + self, mock_tokenizer, mock_request, weather_tool + ): + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool]) + mock_request.tools = [weather_tool] + text = DSML_INVOKE_PREFIX + "made_up_tool" + DSML_INVOKE_NAME_END + " after." + results = simulate_tool_streaming(parser, mock_request, list(text)) + finish_delta = parser.finish_streaming() + + assert collect_function_name(results) is None + content = collect_content(results) + ( + finish_delta.content if finish_delta and finish_delta.content else "" + ) + assert content == text + + def test_quoted_marker_then_wrapped_call_non_streaming( + self, mock_tokenizer, mock_request, weather_tool + ): + """Prose that quotes the invoke marker and never closes it must + not swallow a real wrapped tool call that follows.""" + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool]) + mock_request.tools = [weather_tool] + text = ( + "Docs quote " + + DSML_INVOKE_PREFIX + + " as the marker. " + + _func_calls(_invoke("get_weather", _param("city", "true", "SF"))) + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called + assert len(result.tool_calls) == 1 + assert result.tool_calls[0].function.name == "get_weather" + args = json.loads(result.tool_calls[0].function.arguments) + assert args == {"city": "SF"} + assert DSML_INVOKE_PREFIX in result.content + + def test_quoted_marker_directly_before_wrapped_call( + self, mock_tokenizer, mock_request, weather_tool + ): + """A quoted marker followed immediately by the real wrapper must + release the hold and parse the wrapped call.""" + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool]) + mock_request.tools = [weather_tool] + text = ( + "See " + + DSML_INVOKE_PREFIX + + _func_calls(_invoke("get_weather", _param("city", "true", "SF"))) + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called + assert len(result.tool_calls) == 1 + assert result.tool_calls[0].function.name == "get_weather" + args = json.loads(result.tool_calls[0].function.arguments) + assert args == {"city": "SF"} + assert DSML_INVOKE_PREFIX in result.content + # The wrapper token opens the real tool call, so it must be + # consumed by the parser rather than left in the content. + assert DSML_FUNC_START not in result.content + + def test_streaming_quoted_marker_then_wrapped_call( + self, mock_tokenizer, mock_request, weather_tool + ): + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool]) + mock_request.tools = [weather_tool] + chunks = [ + "Docs quote ", + DSML_INVOKE_PREFIX, + " as the marker. ", + DSML_FUNC_START, + _invoke("get_weather", _param("city", "true", "SF")), + DSML_FUNC_END, + ] + results = simulate_tool_streaming(parser, mock_request, chunks) + + assert collect_function_name(results) == "get_weather" + args = json.loads(collect_tool_arguments(results)) + assert args == {"city": "SF"} + assert DSML_INVOKE_PREFIX in collect_content(results) + + def test_streaming_quoted_marker_prose_released_before_finish( + self, mock_tokenizer, mock_request, weather_tool + ): + """Prose after a quoted marker must stream out promptly instead + of being buffered until the end of the response.""" + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool]) + mock_request.tools = [weather_tool] + prose = "this marker starts a tool call block in the raw output." + chunks = ["Quote: ", DSML_INVOKE_PREFIX, prose, " More prose."] + results = simulate_tool_streaming(parser, mock_request, chunks) + + assert collect_function_name(results) is None + content = collect_content(results) + assert prose in content + assert " More prose." in content + + def test_trailing_prose_after_orphan_invoke_is_kept( + self, mock_tokenizer, mock_request, weather_tool + ): + """A model that drops the opening wrapper often drops the + closing one too, which leaves the response ending between + invokes. The text after the invoke is real output and must + survive as content.""" + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool]) + mock_request.tools = [weather_tool] + text = _invoke("get_weather", _param("city", "true", "SF")) + "\nThanks!" + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called + assert len(result.tool_calls) == 1 + assert result.tool_calls[0].function.name == "get_weather" + assert result.content == "\nThanks!" + + def test_streaming_trailing_prose_after_orphan_invoke_is_kept( + self, mock_tokenizer, mock_request, weather_tool + ): + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool]) + mock_request.tools = [weather_tool] + chunks = [_invoke("get_weather", _param("city", "true", "SF")), "\nThanks!"] + results = simulate_tool_streaming(parser, mock_request, chunks) + + assert collect_function_name(results) == "get_weather" + # Streamed out as it arrives, not buffered until finish. + assert collect_content(results) == "\nThanks!" + + def test_whitespace_between_parallel_orphan_invokes_is_ignored( + self, mock_tokenizer, mock_request, weather_tool + ): + """A response that is only two invokes and the padding between + them comes back with no content at all. + + This case is already covered by the parser dropping content that + is nothing but whitespace when the response called tools, so it + passes whether or not the engine holds the padding back. The + test that actually pins the holding back is + ``test_padding_between_orphan_invokes_is_dropped_after_prose``. + """ + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool, time_tool]) + mock_request.tools = [weather_tool, time_tool] + text = ( + _invoke("get_weather", _param("city", "true", "SF")) + + "\n \n" + + _invoke("get_time", _param("timezone", "true", "EST")) + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called + assert len(result.tool_calls) == 2 + assert result.content is None + + def test_padding_between_orphan_invokes_is_dropped_after_prose( + self, mock_tokenizer, mock_request, weather_tool + ): + """Padding between two recovered invokes is dropped even when + the response already produced real text. + + The prose in front means the content is no longer whitespace + only, so the parser's own whitespace dropping does not apply and + the engine holding the padding back is the only thing keeping it + out. A wrapped call written the same way returns just the + prose, and the recovered call has to match it. + """ + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool, time_tool]) + mock_request.tools = [weather_tool, time_tool] + text = ( + "Some prose " + + _invoke("get_weather", _param("city", "true", "SF")) + + "\n \n" + + _invoke("get_time", _param("timezone", "true", "EST")) + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called + assert len(result.tool_calls) == 2 + assert result.content == "Some prose " + + def test_streaming_padding_between_orphan_invokes_is_dropped_after_prose( + self, mock_tokenizer, mock_request, weather_tool + ): + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool, time_tool]) + mock_request.tools = [weather_tool, time_tool] + chunks = [ + "Some prose ", + _invoke("get_weather", _param("city", "true", "SF")), + "\n \n", + _invoke("get_time", _param("timezone", "true", "EST")), + ] + results = simulate_tool_streaming(parser, mock_request, chunks) + + assert collect_content(results) == "Some prose " + + def test_recovery_does_not_carry_into_a_later_wrapped_call( + self, mock_tokenizer, mock_request, weather_tool + ): + """Once a recovered sequence ends, a later wrapped call in the + same response is treated as an ordinary wrapped call. + + Text between the invokes of a wrapped call is dropped, so if the + engine still thought it was inside a recovered sequence the + stray text below would come back as content. + """ + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool, time_tool]) + mock_request.tools = [weather_tool, time_tool] + text = ( + _invoke("get_weather", _param("city", "true", "SF")) + + DSML_FUNC_END + + DSML_FUNC_START + + _invoke("get_weather", _param("city", "true", "SF")) + + "stray between wrapped invokes" + + _invoke("get_time", _param("timezone", "true", "EST")) + + DSML_FUNC_END + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called + assert len(result.tool_calls) == 3 + assert result.content is None + + def test_streaming_recovery_does_not_carry_into_a_later_wrapped_call( + self, mock_tokenizer, mock_request, weather_tool + ): + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool, time_tool]) + mock_request.tools = [weather_tool, time_tool] + chunks = [ + _invoke("get_weather", _param("city", "true", "SF")), + DSML_FUNC_END, + DSML_FUNC_START, + _invoke("get_weather", _param("city", "true", "SF")), + "stray between wrapped invokes", + _invoke("get_time", _param("timezone", "true", "EST")), + DSML_FUNC_END, + ] + results = simulate_tool_streaming(parser, mock_request, chunks) + + assert collect_content(results) == "" + + def test_padding_held_before_one_invoke_does_not_reach_a_later_gap( + self, mock_tokenizer, mock_request, weather_tool + ): + """Padding held before one invoke is dropped when that invoke + starts, so it cannot reappear in front of later text. + + The first gap is padding and belongs to nothing. Only the + second gap runs into real text, so only that one is content. + """ + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool, time_tool]) + mock_request.tools = [weather_tool, time_tool] + text = ( + _invoke("get_weather", _param("city", "true", "SF")) + + "\n\n" + + _invoke("get_time", _param("timezone", "true", "EST")) + + " " + + "Real text" + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called + assert len(result.tool_calls) == 2 + assert result.content == " Real text" + + def test_streaming_padding_held_before_one_invoke_does_not_reach_a_later_gap( + self, mock_tokenizer, mock_request, weather_tool + ): + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool, time_tool]) + mock_request.tools = [weather_tool, time_tool] + chunks = [ + _invoke("get_weather", _param("city", "true", "SF")), + "\n\n", + _invoke("get_time", _param("timezone", "true", "EST")), + " ", + "Real text", + ] + results = simulate_tool_streaming(parser, mock_request, chunks) + + assert collect_content(results) == " Real text" + + def test_abandoned_recovery_does_not_affect_a_later_wrapped_call( + self, mock_tokenizer, mock_request, weather_tool + ): + """A recovery attempt that turns out not to name a declared tool + must leave nothing behind. + + The invoke below is held while its name is read, then given up + on because ``get_nothing`` was never declared. The wrapped call + after it is ordinary, so the stray text between its invokes is + dropped. + """ + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool, time_tool]) + mock_request.tools = [weather_tool, time_tool] + abandoned = _invoke("get_nothing", _param("city", "true", "SF")) + text = ( + abandoned + + DSML_FUNC_START + + _invoke("get_weather", _param("city", "true", "SF")) + + "stray between wrapped invokes" + + _invoke("get_time", _param("timezone", "true", "EST")) + + DSML_FUNC_END + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called + assert len(result.tool_calls) == 2 + assert result.content == abandoned + + def test_recovery_does_not_leak_into_the_next_request( + self, mock_tokenizer, mock_request, weather_tool + ): + """A response that ends part way through a recovered sequence + must not leave the engine set up for recovery. + + The engine is reused, so without a clean start the next + response would treat an ordinary wrapped call as a recovered + one and hand back the text between its invokes as content. + """ + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool, time_tool]) + mock_request.tools = [weather_tool, time_tool] + + first = parser.extract_tool_calls( + _invoke("get_weather", _param("city", "true", "SF")), mock_request + ) + assert first.tools_called + + second = parser.extract_tool_calls( + DSML_FUNC_START + + _invoke("get_weather", _param("city", "true", "SF")) + + "stray between wrapped invokes" + + _invoke("get_time", _param("timezone", "true", "EST")) + + DSML_FUNC_END, + mock_request, + ) + + assert second.tools_called + assert len(second.tool_calls) == 2 + assert second.content is None + + def test_declared_names_do_not_leak_into_the_next_request( + self, mock_tokenizer, mock_request, weather_tool + ): + """The engine is reused across requests, so a request that + declares no tools must not recover a tool that an earlier + request declared.""" + parser = DeepSeekV32Parser(mock_tokenizer, tools=[weather_tool]) + mock_request.tools = [weather_tool] + text = _invoke("get_weather", _param("city", "true", "SF")) + DSML_FUNC_END + + first = parser.extract_tool_calls(text, mock_request) + assert first.tools_called + + second = parser.extract_tool_calls(text, _request_without_tools()) + + assert not second.tools_called + assert second.tool_calls == [] + assert second.content == text + + # ── Initial state ──────────────────────────────────────────────────── @@ -254,6 +732,20 @@ def test_streaming_wrapper_unwrap_consistency(self, mock_tokenizer, mock_request assert '"arguments"' not in streamed_args assert final_args.startswith(streamed_args) + def test_missing_func_start_orphan_invoke(self, mock_tokenizer, mock_request): + """Orphan invoke without the <|DSML|function_calls> wrapper is + still parsed as a tool call when the request declared the tool + (see gh-48931).""" + tool = _make_tool("get_weather", {"city": {"type": "string"}}) + mock_request.tools = [tool] + text = _invoke("get_weather", _param("city", "true", "SF")) + DSML_FUNC_END + parser = DeepSeekV32Parser(mock_tokenizer, tools=[tool]) + results = simulate_tool_streaming(parser, mock_request, list(text)) + assert collect_function_name(results) == "get_weather" + args = json.loads(collect_tool_arguments(results)) + assert args == {"city": "SF"} + assert "DSML" not in collect_content(results) + def test_missing_invoke_end(self, mock_tokenizer, mock_request): text = ( f"{DSML_FUNC_START}\n" diff --git a/tests/parser/engine/test_deepseek_v4.py b/tests/parser/engine/test_deepseek_v4.py index e5e7b075bac2..5ddb01c5b4b1 100644 --- a/tests/parser/engine/test_deepseek_v4.py +++ b/tests/parser/engine/test_deepseek_v4.py @@ -23,6 +23,7 @@ ) from vllm.parser.abstract_parser import DelegatingParser from vllm.parser.deepseek_v4 import ( + DSML_FOREIGN_TOOL_START, DSML_INVOKE_END, DSML_INVOKE_NAME_END, DSML_INVOKE_PREFIX, @@ -231,6 +232,798 @@ def test_streaming_with_trailing_content(self, mock_tokenizer, mock_request): assert "Done." in collect_content(results) +# ── Missing <|DSML|tool_calls> before <|DSML|invoke ...> ────────── + + +class TestMissingToolStart: + """Orphan invoke blocks are parsed when the START wrapper is missing. + + At long context DeepSeek V4 models intermittently omit the + <|DSML|tool_calls> wrapper while still emitting a well-formed + <|DSML|invoke ...> block. The (CONTENT, INVOKE_PREFIX) transition + recovers the tool call instead of leaking raw DSML into content. + Recovery only accepts tool names the request declared, so these + tests declare the tools they invoke. + See https://github.com/vllm-project/vllm/issues/48931. + """ + + @pytest.fixture + def weather_tool(self): + return _make_tool("get_weather", {"location": {"type": "string"}}) + + def _declared_parser(self, mock_tokenizer, mock_request, *tools): + parser = DeepSeekV4Parser(mock_tokenizer, tools=list(tools)) + mock_request.tools = list(tools) + return parser + + def _orphan_invoke(self, with_tool_end: bool = True) -> str: + text = _invoke("get_weather", ("location", "true", "NYC")) + if with_tool_end: + text += DSML_TOOL_END + return text + + def test_non_streaming_orphan_invoke( + self, mock_tokenizer, mock_request, weather_tool + ): + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + result = parser.extract_tool_calls(self._orphan_invoke(), mock_request) + + assert result.tools_called is True + assert len(result.tool_calls) == 1 + assert result.tool_calls[0].function.name == "get_weather" + args = json.loads(result.tool_calls[0].function.arguments) + assert args == {"location": "NYC"} + assert result.content is None + + def test_non_streaming_orphan_invoke_no_tool_end( + self, mock_tokenizer, mock_request, weather_tool + ): + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + result = parser.extract_tool_calls( + self._orphan_invoke(with_tool_end=False), mock_request + ) + + assert result.tools_called is True + assert len(result.tool_calls) == 1 + assert result.tool_calls[0].function.name == "get_weather" + args = json.loads(result.tool_calls[0].function.arguments) + assert args == {"location": "NYC"} + + def test_non_streaming_orphan_matches_wrapped_parse( + self, mock_tokenizer, mock_request, weather_tool + ): + """The orphan payload parses identically to its wrapped form.""" + invoke = _invoke("get_weather", ("location", "true", "NYC")) + + wrapped_parser = self._declared_parser( + mock_tokenizer, mock_request, weather_tool + ) + wrapped = wrapped_parser.extract_tool_calls( + DSML_TOOL_START + invoke + DSML_TOOL_END, mock_request + ) + orphan_parser = self._declared_parser( + mock_tokenizer, mock_request, weather_tool + ) + orphan = orphan_parser.extract_tool_calls(invoke + DSML_TOOL_END, mock_request) + + assert orphan.tools_called is wrapped.tools_called is True + assert orphan.tool_calls[0].function.name == wrapped.tool_calls[0].function.name + assert ( + orphan.tool_calls[0].function.arguments + == wrapped.tool_calls[0].function.arguments + ) + + def test_streaming_orphan_invoke_split_marker( + self, mock_tokenizer, mock_request, weather_tool + ): + """The invoke marker may arrive split across streaming deltas.""" + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + chunks = [ + "I'll check the weather.\n", + "<|DSML", + '|invoke name="get_weather">', + "\n" + _param("location", "true", "NYC") + "\n", + DSML_INVOKE_END, + DSML_TOOL_END, + ] + + results = simulate_tool_streaming(parser, mock_request, chunks) + + assert collect_function_name(results) == "get_weather" + args = json.loads(collect_tool_arguments(results)) + assert args == {"location": "NYC"} + content = collect_content(results) + assert "I'll check the weather." in content + assert "DSML" not in content + + def test_streaming_orphan_invoke_char_by_char( + self, mock_tokenizer, mock_request, weather_tool + ): + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + text = self._orphan_invoke() + results = simulate_tool_streaming(parser, mock_request, list(text)) + + assert collect_function_name(results) == "get_weather" + args = json.loads(collect_tool_arguments(results)) + assert args == {"location": "NYC"} + assert "DSML" not in collect_content(results) + + def test_orphan_parallel_invokes(self, mock_tokenizer, mock_request, weather_tool): + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = self._declared_parser( + mock_tokenizer, mock_request, weather_tool, time_tool + ) + text = ( + _invoke("get_weather", ("location", "true", "NYC")) + + "\n" + + _invoke("get_time", ("timezone", "true", "EST")) + + DSML_TOOL_END + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is True + assert len(result.tool_calls) == 2 + assert result.tool_calls[0].function.name == "get_weather" + assert result.tool_calls[1].function.name == "get_time" + + def test_plain_content_unaffected(self, mock_tokenizer, mock_request): + parser = DeepSeekV4Parser(mock_tokenizer) + text = 'Use style tags to call tools.' + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is False + assert result.tool_calls == [] + assert result.content == text + + def test_partial_marker_mention_stays_content(self, mock_tokenizer, mock_request): + """A DSML-like fragment that never completes the invoke marker + must be flushed as content, not swallowed.""" + parser = DeepSeekV4Parser(mock_tokenizer) + text = "The prefix <|DSML|invoke is reserved." + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is False + assert result.content == text + + def test_foreign_function_calls_wrapper_still_rejected( + self, mock_tokenizer, mock_request + ): + """An invoke inside the V3.2-style function_calls wrapper stays + plain content: the orphan fallback must not fire inside a + foreign wrapper.""" + parser = DeepSeekV4Parser(mock_tokenizer) + text = _tool_calls( + _invoke("get_weather", ("location", "true", "NYC")), + ).replace("tool_calls", "function_calls") + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is False + assert result.tool_calls == [] + assert result.content == text + + def test_unclosed_foreign_wrapper_then_native_call( + self, mock_tokenizer, mock_request + ): + """A foreign wrapper that never closes must not disable native + tool parsing: the token backed tool_calls wrapper still wins.""" + parser = DeepSeekV4Parser(mock_tokenizer) + text = ( + DSML_FOREIGN_TOOL_START + + "\nStray foreign text.\n" + + _tool_calls(_invoke("get_weather", ("location", "true", "NYC"))) + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is True + assert len(result.tool_calls) == 1 + assert result.tool_calls[0].function.name == "get_weather" + args = json.loads(result.tool_calls[0].function.arguments) + assert args == {"location": "NYC"} + assert "Stray foreign text." in result.content + + def test_orphan_invoke_without_declared_tools_stays_content( + self, mock_tokenizer, mock_request + ): + """A request that declared no tools can never accept a recovered + name, so the orphan invoke stays plain content.""" + parser = DeepSeekV4Parser(mock_tokenizer) + text = self._orphan_invoke() + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is False + assert result.tool_calls == [] + assert result.content == text + + +# ── Orphan invoke name validation ──────────────────────────────────── + + +class TestOrphanInvokeNameValidation: + """Recovered (orphan) invokes must carry a plausible tool name. + + The invoke marker has no dedicated special token in the DeepSeek + vocab, so prose that literally quotes the marker would otherwise be + misparsed as a tool call. The (CONTENT, INVOKE_PREFIX) recovery + transition holds its events until the name completes and only + commits to a tool call when the name is one of the tools the + request declared. The hold ends early once the text seen so far + can no longer grow into a declared tool name, so streaming is not + stalled by prose that quotes the marker. + The wrapped (TOOL_PREAMBLE, INVOKE_PREFIX) path is not validated. + """ + + @pytest.fixture + def weather_tool(self): + return _make_tool("get_weather", {"location": {"type": "string"}}) + + def _declared_parser(self, mock_tokenizer, mock_request, *tools): + parser = DeepSeekV4Parser(mock_tokenizer, tools=list(tools)) + mock_request.tools = list(tools) + return parser + + def test_declared_name_recovered_matches_wrapped( + self, mock_tokenizer, mock_request, weather_tool + ): + invoke = _invoke("get_weather", ("location", "true", "NYC")) + + wrapped_parser = self._declared_parser( + mock_tokenizer, mock_request, weather_tool + ) + wrapped = wrapped_parser.extract_tool_calls( + DSML_TOOL_START + invoke + DSML_TOOL_END, mock_request + ) + orphan_parser = self._declared_parser( + mock_tokenizer, mock_request, weather_tool + ) + orphan = orphan_parser.extract_tool_calls(invoke + DSML_TOOL_END, mock_request) + + assert orphan.tools_called is wrapped.tools_called is True + assert ( + orphan.tool_calls[0].function.name + == wrapped.tool_calls[0].function.name + == "get_weather" + ) + assert ( + orphan.tool_calls[0].function.arguments + == wrapped.tool_calls[0].function.arguments + ) + + def test_undeclared_name_stays_content( + self, mock_tokenizer, mock_request, weather_tool + ): + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + text = ( + "The marker " + + DSML_INVOKE_PREFIX + + "made_up_tool" + + DSML_INVOKE_NAME_END + + " is reserved syntax." + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is False + assert result.tool_calls == [] + assert result.content == text + + def test_undeclared_orphan_then_wrapped_call_still_parses( + self, mock_tokenizer, mock_request, weather_tool + ): + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + text = ( + DSML_INVOKE_PREFIX + + "made_up_tool" + + DSML_INVOKE_NAME_END + + " then a real call: " + + _tool_calls(_invoke("get_weather", ("location", "true", "NYC"))) + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is True + assert len(result.tool_calls) == 1 + assert result.tool_calls[0].function.name == "get_weather" + args = json.loads(result.tool_calls[0].function.arguments) + assert args == {"location": "NYC"} + assert DSML_INVOKE_PREFIX + "made_up_tool" in result.content + + def test_no_tools_name_with_space_stays_content(self, mock_tokenizer, mock_request): + parser = DeepSeekV4Parser(mock_tokenizer) + text = DSML_INVOKE_PREFIX + "not a name" + DSML_INVOKE_NAME_END + " more text." + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is False + assert result.content == text + + def test_no_tools_empty_name_stays_content(self, mock_tokenizer, mock_request): + parser = DeepSeekV4Parser(mock_tokenizer) + text = DSML_INVOKE_PREFIX + DSML_INVOKE_NAME_END + " more text." + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is False + assert result.content == text + + def test_truncated_name_flushes_content( + self, mock_tokenizer, mock_request, weather_tool + ): + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + text = "Say " + DSML_INVOKE_PREFIX + "get_wea" + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is False + assert result.tool_calls == [] + assert result.content == text + + def test_streaming_ends_mid_name_flushes_content( + self, mock_tokenizer, mock_request, weather_tool + ): + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + chunks = ["Say ", DSML_INVOKE_PREFIX, "get_wea"] + results = simulate_tool_streaming(parser, mock_request, chunks) + finish_delta = parser.finish_streaming() + + assert collect_function_name(results) is None + assert finish_delta is not None + assert not finish_delta.tool_calls + content = collect_content(results) + (finish_delta.content or "") + assert content == "Say " + DSML_INVOKE_PREFIX + "get_wea" + + def test_char_by_char_declared_name_recovers( + self, mock_tokenizer, mock_request, weather_tool + ): + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + text = _invoke("get_weather", ("location", "true", "NYC")) + DSML_TOOL_END + results = simulate_tool_streaming(parser, mock_request, list(text)) + + assert collect_function_name(results) == "get_weather" + args = json.loads(collect_tool_arguments(results)) + assert args == {"location": "NYC"} + assert "DSML" not in collect_content(results) + + def test_char_by_char_undeclared_name_stays_content( + self, mock_tokenizer, mock_request, weather_tool + ): + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + text = DSML_INVOKE_PREFIX + "made_up_tool" + DSML_INVOKE_NAME_END + " after." + results = simulate_tool_streaming(parser, mock_request, list(text)) + finish_delta = parser.finish_streaming() + + assert collect_function_name(results) is None + content = collect_content(results) + ( + finish_delta.content if finish_delta and finish_delta.content else "" + ) + assert content == text + + def test_wrapped_path_not_validated( + self, mock_tokenizer, mock_request, weather_tool + ): + """An undeclared name inside the tool_calls wrapper still parses: + validation applies only to the orphan recovery path.""" + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + text = _tool_calls(_invoke("undeclared_fn", ("location", "true", "NYC"))) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is True + assert len(result.tool_calls) == 1 + assert result.tool_calls[0].function.name == "undeclared_fn" + + def test_wrapped_path_never_holds_events( + self, mock_tokenizer, mock_request, weather_tool + ): + """TOOL_CALL_START fires immediately on the wrapped path, before + the name completes: no hold window and no tool_index rewind.""" + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + simulate_tool_streaming( + parser, mock_request, [DSML_TOOL_START, DSML_INVOKE_PREFIX] + ) + engine = parser._engine + + assert engine._hold_active is False + assert engine.tool_index == 0 + + def test_parallel_orphan_invokes_with_declared_tools( + self, mock_tokenizer, mock_request, weather_tool + ): + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = self._declared_parser( + mock_tokenizer, mock_request, weather_tool, time_tool + ) + text = ( + _invoke("get_weather", ("location", "true", "NYC")) + + "\n" + + _invoke("get_time", ("timezone", "true", "EST")) + + DSML_TOOL_END + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is True + assert len(result.tool_calls) == 2 + assert result.tool_calls[0].function.name == "get_weather" + assert result.tool_calls[1].function.name == "get_time" + + def test_quoted_marker_then_wrapped_call_non_streaming( + self, mock_tokenizer, mock_request, weather_tool + ): + """Prose that quotes the invoke marker and never closes it must + not swallow a real wrapped tool call that follows.""" + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + text = ( + "Docs quote " + + DSML_INVOKE_PREFIX + + " as the marker. " + + _tool_calls(_invoke("get_weather", ("location", "true", "NYC"))) + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is True + assert len(result.tool_calls) == 1 + assert result.tool_calls[0].function.name == "get_weather" + args = json.loads(result.tool_calls[0].function.arguments) + assert args == {"location": "NYC"} + assert DSML_INVOKE_PREFIX in result.content + + def test_quoted_marker_directly_before_wrapped_call( + self, mock_tokenizer, mock_request, weather_tool + ): + """A quoted marker followed immediately by the real wrapper must + release the hold and parse the wrapped call.""" + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + text = ( + "See " + + DSML_INVOKE_PREFIX + + _tool_calls(_invoke("get_weather", ("location", "true", "NYC"))) + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is True + assert len(result.tool_calls) == 1 + assert result.tool_calls[0].function.name == "get_weather" + args = json.loads(result.tool_calls[0].function.arguments) + assert args == {"location": "NYC"} + assert DSML_INVOKE_PREFIX in result.content + # The wrapper token opens the real tool call, so it must be + # consumed by the parser rather than left in the content. + assert DSML_TOOL_START not in result.content + + def test_streaming_quoted_marker_then_wrapped_call( + self, mock_tokenizer, mock_request, weather_tool + ): + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + chunks = [ + "Docs quote ", + DSML_INVOKE_PREFIX, + " as the marker. ", + DSML_TOOL_START, + _invoke("get_weather", ("location", "true", "NYC")), + DSML_TOOL_END, + ] + results = simulate_tool_streaming(parser, mock_request, chunks) + + assert collect_function_name(results) == "get_weather" + args = json.loads(collect_tool_arguments(results)) + assert args == {"location": "NYC"} + content = collect_content(results) + assert DSML_INVOKE_PREFIX in content + + def test_streaming_quoted_marker_prose_released_before_finish( + self, mock_tokenizer, mock_request, weather_tool + ): + """Prose after a quoted marker must stream out promptly instead + of being buffered until the end of the response.""" + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + prose = "this marker starts a tool call block in the raw output." + chunks = ["Quote: ", DSML_INVOKE_PREFIX, prose, " More prose."] + results = simulate_tool_streaming(parser, mock_request, chunks) + + assert collect_function_name(results) is None + content = collect_content(results) + assert prose in content + assert " More prose." in content + + def test_tool_choice_none_keeps_quoted_invoke_as_content( + self, mock_tokenizer, mock_request, weather_tool + ): + """With tool_choice set to none, invoke text must stay in the + content instead of being consumed by tool recovery.""" + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + mock_request.tool_choice = "none" + text = ( + "Docs say you write " + + _invoke("get_weather", ("location", "true", "NYC")) + + " to call it." + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is False + assert result.tool_calls == [] + assert result.content == text + + def test_non_ascii_declared_name_recovered(self, mock_tokenizer, mock_request): + """A declared tool name is recoverable even when it contains + characters outside the ASCII range.""" + tool = _make_tool("查询天气", {"location": {"type": "string"}}) + parser = self._declared_parser(mock_tokenizer, mock_request, tool) + text = _invoke("查询天气", ("location", "true", "NYC")) + DSML_TOOL_END + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is True + assert len(result.tool_calls) == 1 + assert result.tool_calls[0].function.name == "查询天气" + args = json.loads(result.tool_calls[0].function.arguments) + assert args == {"location": "NYC"} + + def test_trailing_prose_after_orphan_invoke_is_kept( + self, mock_tokenizer, mock_request, weather_tool + ): + """A model that drops the opening wrapper often drops the + closing one too, which leaves the response ending between + invokes. The text after the invoke is real output and must + survive as content.""" + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + text = _invoke("get_weather", ("location", "true", "NYC")) + "\nThanks!" + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is True + assert len(result.tool_calls) == 1 + assert result.tool_calls[0].function.name == "get_weather" + assert result.content == "\nThanks!" + + def test_streaming_trailing_prose_after_orphan_invoke_is_kept( + self, mock_tokenizer, mock_request, weather_tool + ): + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + chunks = [_invoke("get_weather", ("location", "true", "NYC")), "\nThanks!"] + results = simulate_tool_streaming(parser, mock_request, chunks) + + assert collect_function_name(results) == "get_weather" + # Streamed out as it arrives, not buffered until finish. + assert collect_content(results) == "\nThanks!" + + def test_whitespace_between_parallel_orphan_invokes_is_ignored( + self, mock_tokenizer, mock_request, weather_tool + ): + """A response that is only two invokes and the padding between + them comes back with no content at all. + + This case is already covered by the parser dropping content that + is nothing but whitespace when the response called tools, so it + passes whether or not the engine holds the padding back. The + test that actually pins the holding back is + ``test_padding_between_orphan_invokes_is_dropped_after_prose``. + """ + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = self._declared_parser( + mock_tokenizer, mock_request, weather_tool, time_tool + ) + text = ( + _invoke("get_weather", ("location", "true", "NYC")) + + "\n \n" + + _invoke("get_time", ("timezone", "true", "EST")) + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is True + assert len(result.tool_calls) == 2 + assert result.content is None + + def test_padding_between_orphan_invokes_is_dropped_after_prose( + self, mock_tokenizer, mock_request, weather_tool + ): + """Padding between two recovered invokes is dropped even when + the response already produced real text. + + The prose in front means the content is no longer whitespace + only, so the parser's own whitespace dropping does not apply and + the engine holding the padding back is the only thing keeping it + out. A wrapped call written the same way returns just the + prose, and the recovered call has to match it. + """ + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = self._declared_parser( + mock_tokenizer, mock_request, weather_tool, time_tool + ) + text = ( + "Some prose " + + _invoke("get_weather", ("location", "true", "NYC")) + + "\n \n" + + _invoke("get_time", ("timezone", "true", "EST")) + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is True + assert len(result.tool_calls) == 2 + assert result.content == "Some prose " + + def test_streaming_padding_between_orphan_invokes_is_dropped_after_prose( + self, mock_tokenizer, mock_request, weather_tool + ): + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = self._declared_parser( + mock_tokenizer, mock_request, weather_tool, time_tool + ) + chunks = [ + "Some prose ", + _invoke("get_weather", ("location", "true", "NYC")), + "\n \n", + _invoke("get_time", ("timezone", "true", "EST")), + ] + results = simulate_tool_streaming(parser, mock_request, chunks) + + assert collect_content(results) == "Some prose " + + def test_recovery_does_not_carry_into_a_later_wrapped_call( + self, mock_tokenizer, mock_request, weather_tool + ): + """Once a recovered sequence ends, a later wrapped call in the + same response is treated as an ordinary wrapped call. + + Text between the invokes of a wrapped call is dropped, so if the + engine still thought it was inside a recovered sequence the + stray text below would come back as content. + """ + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = self._declared_parser( + mock_tokenizer, mock_request, weather_tool, time_tool + ) + text = ( + _invoke("get_weather", ("location", "true", "NYC")) + + DSML_TOOL_END + + DSML_TOOL_START + + _invoke("get_weather", ("location", "true", "NYC")) + + "stray between wrapped invokes" + + _invoke("get_time", ("timezone", "true", "EST")) + + DSML_TOOL_END + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is True + assert len(result.tool_calls) == 3 + assert result.content is None + + def test_streaming_recovery_does_not_carry_into_a_later_wrapped_call( + self, mock_tokenizer, mock_request, weather_tool + ): + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = self._declared_parser( + mock_tokenizer, mock_request, weather_tool, time_tool + ) + chunks = [ + _invoke("get_weather", ("location", "true", "NYC")), + DSML_TOOL_END, + DSML_TOOL_START, + _invoke("get_weather", ("location", "true", "NYC")), + "stray between wrapped invokes", + _invoke("get_time", ("timezone", "true", "EST")), + DSML_TOOL_END, + ] + results = simulate_tool_streaming(parser, mock_request, chunks) + + assert collect_content(results) == "" + + def test_padding_held_before_one_invoke_does_not_reach_a_later_gap( + self, mock_tokenizer, mock_request, weather_tool + ): + """Padding held before one invoke is dropped when that invoke + starts, so it cannot reappear in front of later text. + + The first gap is padding and belongs to nothing. Only the + second gap runs into real text, so only that one is content. + """ + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = self._declared_parser( + mock_tokenizer, mock_request, weather_tool, time_tool + ) + text = ( + _invoke("get_weather", ("location", "true", "NYC")) + + "\n\n" + + _invoke("get_time", ("timezone", "true", "EST")) + + " " + + "Real text" + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is True + assert len(result.tool_calls) == 2 + assert result.content == " Real text" + + def test_streaming_padding_held_before_one_invoke_does_not_reach_a_later_gap( + self, mock_tokenizer, mock_request, weather_tool + ): + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = self._declared_parser( + mock_tokenizer, mock_request, weather_tool, time_tool + ) + chunks = [ + _invoke("get_weather", ("location", "true", "NYC")), + "\n\n", + _invoke("get_time", ("timezone", "true", "EST")), + " ", + "Real text", + ] + results = simulate_tool_streaming(parser, mock_request, chunks) + + assert collect_content(results) == " Real text" + + def test_abandoned_recovery_does_not_affect_a_later_wrapped_call( + self, mock_tokenizer, mock_request, weather_tool + ): + """A recovery attempt that turns out not to name a declared tool + must leave nothing behind. + + The invoke below is held while its name is read, then given up + on because ``get_nothing`` was never declared. The wrapped call + after it is ordinary, so the stray text between its invokes is + dropped. + """ + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = self._declared_parser( + mock_tokenizer, mock_request, weather_tool, time_tool + ) + abandoned = _invoke("get_nothing", ("location", "true", "NYC")) + text = ( + abandoned + + DSML_TOOL_START + + _invoke("get_weather", ("location", "true", "NYC")) + + "stray between wrapped invokes" + + _invoke("get_time", ("timezone", "true", "EST")) + + DSML_TOOL_END + ) + result = parser.extract_tool_calls(text, mock_request) + + assert result.tools_called is True + assert len(result.tool_calls) == 2 + assert result.content == abandoned + + def test_recovery_does_not_leak_into_the_next_request( + self, mock_tokenizer, mock_request, weather_tool + ): + """A response that ends part way through a recovered sequence + must not leave the engine set up for recovery. + + The engine is reused, so without a clean start the next + response would treat an ordinary wrapped call as a recovered + one and hand back the text between its invokes as content. + """ + time_tool = _make_tool("get_time", {"timezone": {"type": "string"}}) + parser = self._declared_parser( + mock_tokenizer, mock_request, weather_tool, time_tool + ) + + first = parser.extract_tool_calls( + _invoke("get_weather", ("location", "true", "NYC")), mock_request + ) + assert first.tools_called is True + + second = parser.extract_tool_calls( + DSML_TOOL_START + + _invoke("get_weather", ("location", "true", "NYC")) + + "stray between wrapped invokes" + + _invoke("get_time", ("timezone", "true", "EST")) + + DSML_TOOL_END, + mock_request, + ) + + assert second.tools_called is True + assert len(second.tool_calls) == 2 + assert second.content is None + + def test_declared_names_do_not_leak_into_the_next_request( + self, mock_tokenizer, mock_request, weather_tool + ): + """The engine is reused across requests, so a request that + declares no tools must not recover a tool that an earlier + request declared.""" + parser = self._declared_parser(mock_tokenizer, mock_request, weather_tool) + text = _invoke("get_weather", ("location", "true", "NYC")) + DSML_TOOL_END + + first = parser.extract_tool_calls(text, mock_request) + assert first.tools_called is True + + second = parser.extract_tool_calls(text, _request_without_tools()) + + assert second.tools_called is False + assert second.tool_calls == [] + assert second.content == text + + # ── Thinking mode initial state ────────────────────────────────────── @@ -540,6 +1333,20 @@ def _tool_calls(*invokes): return DSML_TOOL_START + "\n".join(invokes) + DSML_TOOL_END +def _request_without_tools(): + from unittest.mock import MagicMock + + from vllm.entrypoints.openai.chat_completion.protocol import ( # noqa: E501 + ChatCompletionRequest, + ) + + req = MagicMock(spec=ChatCompletionRequest) + req.tools = [] + req.tool_choice = "auto" + req.include_reasoning = True + return req + + class TestParallelUnwrapping: @pytest.fixture def weather_tool(self): diff --git a/vllm/parser/deepseek_v32.py b/vllm/parser/deepseek_v32.py index 0d9ac9f53ce1..30768d660ffe 100644 --- a/vllm/parser/deepseek_v32.py +++ b/vllm/parser/deepseek_v32.py @@ -26,6 +26,8 @@ DSML_INVOKE_NAME_END, DSML_INVOKE_PREFIX, DSML_PARAM_CLOSE, + DSML_TOOL_END, + DSML_TOOL_START, _dsml_arg_converter, _unwrap_wrapper_args, ) @@ -59,6 +61,8 @@ def deepseek_v32_config() -> ParserEngineConfig: "INVOKE_NAME_END": DSML_INVOKE_NAME_END, "INVOKE_END": DSML_INVOKE_END, "PARAM_CLOSE": DSML_PARAM_CLOSE, + "FOREIGN_START": DSML_TOOL_START, + "FOREIGN_END": DSML_TOOL_END, }, token_id_terminals={ "TOOL_START": DSML_FUNC_START, @@ -69,6 +73,34 @@ def deepseek_v32_config() -> ParserEngineConfig: ParserState.TOOL_PREAMBLE, (), ), + # Orphan invoke: at long context the model may omit the + # <|DSML|function_calls> wrapper and emit the invoke + # directly. The invoke marker has no dedicated special + # token, so hold events and validate the parsed name + # before committing. Only names the request declared are + # accepted. + (ParserState.CONTENT, "INVOKE_PREFIX"): Transition( + ParserState.TOOL_NAME, + (EventType.TOOL_CALL_START,), + validate_tool_name=True, + ), + # V4-style tool_calls wrapper is foreign to V3.2: pass it + # and its contents through as plain content + (ParserState.CONTENT, "FOREIGN_START"): Transition( + ParserState.FOREIGN_BLOCK, + (EventType.TEXT_CHUNK,), + ), + (ParserState.FOREIGN_BLOCK, "FOREIGN_END"): Transition( + ParserState.CONTENT, + (EventType.TEXT_CHUNK,), + ), + # The native wrapper always wins over an unclosed foreign + # block, so a stray foreign start cannot disable tool + # parsing for the rest of the response. + (ParserState.FOREIGN_BLOCK, "TOOL_START"): Transition( + ParserState.TOOL_PREAMBLE, + (), + ), (ParserState.TOOL_PREAMBLE, "INVOKE_PREFIX"): Transition( ParserState.TOOL_NAME, (EventType.TOOL_CALL_START,), @@ -99,6 +131,7 @@ def deepseek_v32_config() -> ParserEngineConfig: ParserState.CONTENT: EventType.TEXT_CHUNK, ParserState.TOOL_NAME: EventType.TOOL_NAME, ParserState.TOOL_ARGS: EventType.ARG_VALUE_CHUNK, + ParserState.FOREIGN_BLOCK: EventType.TEXT_CHUNK, }, arg_converter=_dsml_arg_converter, arg_structural_chars=frozenset(">"), diff --git a/vllm/parser/deepseek_v4.py b/vllm/parser/deepseek_v4.py index c65668708202..94c5b83c4fcb 100644 --- a/vllm/parser/deepseek_v4.py +++ b/vllm/parser/deepseek_v4.py @@ -48,6 +48,9 @@ DSML_INVOKE_NAME_END = '">' DSML_INVOKE_END = f"" DSML_PARAM_CLOSE = f"" +# DeepSeek V3.2-style wrapper, recognized only to reject it as foreign +DSML_FOREIGN_TOOL_START = f"<{_DSML}function_calls>" +DSML_FOREIGN_TOOL_END = f"" _ESCAPED_DSML = re.escape(_DSML) _PARAM_RE = re.compile( @@ -135,6 +138,8 @@ def deepseek_v4_config(thinking: bool = False) -> ParserEngineConfig: "INVOKE_NAME_END": DSML_INVOKE_NAME_END, "INVOKE_END": DSML_INVOKE_END, "PARAM_CLOSE": DSML_PARAM_CLOSE, + "FOREIGN_START": DSML_FOREIGN_TOOL_START, + "FOREIGN_END": DSML_FOREIGN_TOOL_END, }, token_id_terminals={ "THINK_START": DSML_THINK_START, @@ -170,6 +175,33 @@ def deepseek_v4_config(thinking: bool = False) -> ParserEngineConfig: ParserState.TOOL_PREAMBLE, (), ), + # Orphan invoke: at long context the model may omit the + # <|DSML|tool_calls> wrapper and emit the invoke directly. + # The invoke marker has no dedicated special token, so hold + # events and validate the parsed name before committing. + # Only names the request declared are accepted. + (ParserState.CONTENT, "INVOKE_PREFIX"): Transition( + ParserState.TOOL_NAME, + (EventType.TOOL_CALL_START,), + validate_tool_name=True, + ), + # V3.2-style function_calls wrapper is foreign to V4: pass + # it and its contents through as plain content + (ParserState.CONTENT, "FOREIGN_START"): Transition( + ParserState.FOREIGN_BLOCK, + (EventType.TEXT_CHUNK,), + ), + (ParserState.FOREIGN_BLOCK, "FOREIGN_END"): Transition( + ParserState.CONTENT, + (EventType.TEXT_CHUNK,), + ), + # The native wrapper always wins over an unclosed foreign + # block, so a stray foreign start cannot disable tool + # parsing for the rest of the response. + (ParserState.FOREIGN_BLOCK, "TOOL_START"): Transition( + ParserState.TOOL_PREAMBLE, + (), + ), (ParserState.TOOL_PREAMBLE, "INVOKE_PREFIX"): Transition( ParserState.TOOL_NAME, (EventType.TOOL_CALL_START,), @@ -201,6 +233,7 @@ def deepseek_v4_config(thinking: bool = False) -> ParserEngineConfig: ParserState.REASONING: EventType.REASONING_CHUNK, ParserState.TOOL_NAME: EventType.TOOL_NAME, ParserState.TOOL_ARGS: EventType.ARG_VALUE_CHUNK, + ParserState.FOREIGN_BLOCK: EventType.TEXT_CHUNK, }, arg_converter=_dsml_arg_converter, arg_structural_chars=frozenset(">"), diff --git a/vllm/parser/engine/parser_engine.py b/vllm/parser/engine/parser_engine.py index 048e714cb4ae..b58405695fc4 100644 --- a/vllm/parser/engine/parser_engine.py +++ b/vllm/parser/engine/parser_engine.py @@ -29,6 +29,7 @@ from vllm.parser.engine.streaming_parser_engine import StreamingParserEngine from vllm.tool_parsers.utils import ( coerce_to_schema_type, + collect_tool_names, extract_types_from_schema, find_tool_name, find_tool_properties, @@ -107,6 +108,7 @@ def __init__( self._engine = StreamingParserEngine( parser_engine_config, tokenizer, vocab=self.vocab ) + self._engine.allowed_tool_names = self._declared_tool_names() self._has_reasoning = ( "THINK_END" in parser_engine_config.token_id_terminals @@ -401,6 +403,11 @@ def _accept_tool_name(self, name: str) -> bool: # ── Private helpers ───────────────────────────────────────────── + def _declared_tool_names(self) -> frozenset[str] | None: + if not self._tools: + return None + return collect_tool_names(self._tools) or None + def _check_skip_tool_parsing( self, request: ChatCompletionRequest | ResponsesRequest, @@ -408,10 +415,21 @@ def _check_skip_tool_parsing( tools = getattr(request, "tools", None) if tools: self._tools = tools + self._engine.allowed_tool_names = self._declared_tool_names() + else: + # The engine is reused across requests and reset() keeps this + # field, so it has to be cleared here. Otherwise a request + # that declares no tools would inherit the names of the + # previous one and could recover a tool it never asked for. + self._engine.allowed_tool_names = None if not self.skip_tool_parsing and not self._suppress_tool_calls: tool_choice = getattr(request, "tool_choice", None) if tool_choice == "none" and tools: self._suppress_tool_calls = True + # The engine needs the suppression state too: recovery + # transitions must not consume text that will never be allowed + # to become a tool call. + self._engine.suppress_tool_calls = self._suppress_tool_calls def _strip_content_whitespace( self, diff --git a/vllm/parser/engine/parser_engine_config.py b/vllm/parser/engine/parser_engine_config.py index ad83e331490a..196256706ead 100644 --- a/vllm/parser/engine/parser_engine_config.py +++ b/vllm/parser/engine/parser_engine_config.py @@ -33,6 +33,9 @@ class ParserState(Enum): TOOL_NAME = auto() TOOL_ARGS = auto() TOOL_BETWEEN = auto() + # Inside a block belonging to a different model format; terminals + # matched here pass through as plain content. + FOREIGN_BLOCK = auto() @dataclass(frozen=True, slots=True) @@ -40,6 +43,11 @@ class Transition: next_state: ParserState events: tuple[EventType, ...] = field(default_factory=tuple) skip_in_token_id_mode: bool = False + # Hold this transition's events until the tool name completes, then + # validate the name before committing to the tool call. Set on + # recovery transitions whose trigger marker has no dedicated special + # token, so prose quoting the marker is not misparsed as a tool call. + validate_tool_name: bool = False @dataclass(frozen=True) diff --git a/vllm/parser/engine/streaming_parser_engine.py b/vllm/parser/engine/streaming_parser_engine.py index f5c395f76d3f..dbccb73cda3c 100644 --- a/vllm/parser/engine/streaming_parser_engine.py +++ b/vllm/parser/engine/streaming_parser_engine.py @@ -158,6 +158,17 @@ def __init__( ) self.skip_tool_parsing = False + # Function names declared by the request, or None when unknown. + # Consulted only by transitions with ``validate_tool_name``; + # set per request by the owning ParserEngine, like + # ``skip_tool_parsing`` it survives reset(). + self.allowed_tool_names: frozenset[str] | None = None + # True when the request asked for tool_choice "none". Recovery + # transitions are skipped while set, so text that looks like a + # recovered tool call stays plain content instead of being + # consumed and then suppressed. Set per request by the owning + # ParserEngine; survives reset() like ``skip_tool_parsing``. + self.suppress_tool_calls = False self.reset(initial_state=initial_state) def _reset_args_state(self) -> None: @@ -187,6 +198,14 @@ def reset(self, initial_state: ParserState | None = None) -> None: self._lexer.reset() self._message_header_buffer = "" self._reset_args_state() + self._recovered_tool_call = False + self._pending_between_text = "" + self._hold_active = False + self._held_events: list[SemanticEvent] = [] + self._held_raw: list[str] = [] + self._held_name: list[str] = [] + self._held_prior_state: ParserState = self.state + self._held_prior_tool_index: int = -1 def feed( self, @@ -240,6 +259,12 @@ def finish(self) -> list[SemanticEvent]: events.extend(self._process_lex_tokens(self._lexer.flush())) + if self._hold_active: + # Stream ended before the recovered tool name completed: + # the held events never validated, so flush the raw text + # as content in the pre-recovery state. + events.extend(self._abort_hold("".join(self._held_raw))) + if self._args_buffer: events.append( SemanticEvent( @@ -316,6 +341,15 @@ def _on_terminal(self, terminal: str, value: str) -> list[SemanticEvent]: if transition is None: if self._has_drops and terminal == DROP_TERMINAL: return [] + if self._hold_active and self.state == ParserState.TOOL_NAME: + # A terminal with no meaning inside a held tool name, + # for example a real tool call start token, ends the + # hold: replay the held text as content, then handle + # the terminal again in the restored state so it keeps + # its normal meaning. + events = self._abort_hold("".join(self._held_raw)) + events.extend(self._on_terminal(terminal, value)) + return events return self._emit_for_state(value) if self.skip_tool_parsing and terminal in self._tool_terminals: @@ -356,6 +390,23 @@ def _on_terminal(self, terminal: str, value: str) -> list[SemanticEvent]: return self._apply_transition(transition, value) def _emit_for_state(self, text: str) -> list[SemanticEvent]: + if self._hold_active and self.state == ParserState.TOOL_NAME: + candidate = "".join(self._held_name) + text + if not self._can_grow_into_declared_name(candidate): + # The held text can no longer become a declared tool + # name, so holding longer would only stall streaming. + # Release everything consumed so far as content. + return self._abort_hold("".join(self._held_raw) + text) + self._held_raw.append(text) + self._held_name.append(text) + self._held_events.append( + SemanticEvent( + EventType.TOOL_NAME, + value=text, + tool_index=self.tool_index, + ) + ) + return [] if self.state == ParserState.MESSAGE_HEADER: self._message_header_buffer += text return [] @@ -372,6 +423,24 @@ def _emit_for_state(self, text: str) -> list[SemanticEvent]: content_type = self.config.content_events.get(self.state) if content_type is not None: return [SemanticEvent(content_type, value=text, tool_index=self.tool_index)] + if self._recovered_tool_call and self.state == ParserState.TOOL_BETWEEN: + # A response that lost its opening wrapper usually loses the + # closing one too, so text after a recovered invoke is often + # the rest of the answer rather than padding before the next + # invoke. Whitespace is held back because that is what + # padding looks like; as soon as anything else shows up the + # whole run is real output and goes out as content. + self._pending_between_text += text + if self._pending_between_text.strip(): + held = self._pending_between_text + self._pending_between_text = "" + return [ + SemanticEvent( + EventType.TEXT_CHUNK, + value=held, + tool_index=self.tool_index, + ) + ] return [] def _on_content(self, text: str) -> list[SemanticEvent]: @@ -383,6 +452,87 @@ def _apply_transition( self, transition: Transition, value: str, + ) -> list[SemanticEvent]: + if self._hold_active: + return self._resolve_hold(transition, value) + if transition.validate_tool_name: + if self.suppress_tool_calls or self.allowed_tool_names is None: + # Recovery could never be accepted for this request, so + # the trigger text stays plain content and nothing is + # buffered. + return self._emit_for_state(value) + return self._begin_hold(transition, value) + return self._run_transition(transition, value) + + def _begin_hold( + self, + transition: Transition, + value: str, + ) -> list[SemanticEvent]: + """Apply a ``validate_tool_name`` transition but hold its events. + + The events (and every TOOL_NAME chunk that follows) stay + buffered until the name completes and validates, so a false + positive can be undone without having emitted anything. + """ + prior_state = self.state + prior_tool_index = self.tool_index + self._held_events = self._run_transition(transition, value) + self._held_raw = [value] + self._held_name = [] + self._held_prior_state = prior_state + self._held_prior_tool_index = prior_tool_index + self._hold_active = True + self._recovered_tool_call = True + return [] + + def _resolve_hold( + self, + transition: Transition, + value: str, + ) -> list[SemanticEvent]: + """End the hold window at the name-completing transition.""" + name = "".join(self._held_name) + allowed = self.allowed_tool_names + if allowed is not None and name in allowed: + events = self._held_events + self._clear_hold() + events.extend(self._run_transition(transition, value)) + return events + return self._abort_hold("".join(self._held_raw) + value) + + def _abort_hold(self, raw: str) -> list[SemanticEvent]: + """Discard held events and re-emit the raw text as content.""" + self.state = self._held_prior_state + self.tool_index = self._held_prior_tool_index + self._recovered_tool_call = self._held_prior_state in self._TOOL_STATES + self._clear_hold() + return self._emit_for_state(raw) + + def _clear_hold(self) -> None: + self._hold_active = False + self._held_events = [] + self._held_raw = [] + self._held_name = [] + + def _can_grow_into_declared_name(self, candidate: str) -> bool: + """Return True when *candidate* is a prefix of a declared tool name. + + Consulted while a recovery hold is active. Membership in the + declared set is the only way a held name can validate, so once + the text seen so far stops being a prefix of any declared name + the caller aborts the hold. This also bounds how much text a + hold can buffer to the length of the longest declared name. + """ + allowed = self.allowed_tool_names + if allowed is None: + return False + return any(name.startswith(candidate) for name in allowed) + + def _run_transition( + self, + transition: Transition, + value: str, ) -> list[SemanticEvent]: events: list[SemanticEvent] = [] previous_state = self.state @@ -402,10 +552,17 @@ def _apply_transition( ) self._args_buffer = "" + # Whatever is still held between invokes is whitespace padding, + # which the wrapped path drops too. + self._pending_between_text = "" + if previous_state == ParserState.MESSAGE_HEADER: message_header = self._message_header_buffer self._message_header_buffer = "" + if transition.next_state not in self._TOOL_STATES: + self._recovered_tool_call = False + self.state = transition.next_state for event_type in transition.events: diff --git a/vllm/tool_parsers/utils.py b/vllm/tool_parsers/utils.py index 95769bafd7f3..290004fdbb22 100644 --- a/vllm/tool_parsers/utils.py +++ b/vllm/tool_parsers/utils.py @@ -289,6 +289,23 @@ def find_tool_properties( return {} +def collect_tool_names(tools: list[Tool] | None) -> frozenset[str]: + """Collect the names of all declared function tools.""" + if not tools: + return frozenset() + names: set[str] = set() + for tool in tools: + if isinstance(tool, (FunctionTool, NamespaceTool)): + for name, _ in iter_response_function_tool_info(tool): + names.add(name) + continue + if not _is_function_tool(tool): + continue + name, _ = _extract_tool_info(tool) + names.add(name) + return frozenset(names) + + def find_tool_name( tools: list[Tool] | None, tool_name: str,