diff --git a/python/sglang/srt/function_call/glm47_moe_detector.py b/python/sglang/srt/function_call/glm47_moe_detector.py index 9bc11d7703e8..05b9c5413625 100644 --- a/python/sglang/srt/function_call/glm47_moe_detector.py +++ b/python/sglang/srt/function_call/glm47_moe_detector.py @@ -412,7 +412,28 @@ def _process_xml_to_json_streaming( ) and closing_tag.startswith(self._xml_tag_buffer) if not is_potential_closing: - content = self._xml_tag_buffer + # The buffer is not a prefix of "", but a + # proper suffix of it might still be. Slide forward to + # the smallest k > 0 such that buffer[k:] is a prefix + # of the closing tag and keep that suffix buffered; + # everything before it is real value content. Without + # this, a value ending with "<" (e.g. ".<") followed + # immediately by "" would yield buffer + # "<<", get ejected wholesale, and the actual closing + # tag would never match — leaking "" into + # the emitted JSON arguments. + kept_suffix = "" + for k in range(1, len(self._xml_tag_buffer)): + candidate = self._xml_tag_buffer[k:] + if closing_tag.startswith(candidate): + kept_suffix = candidate + break + content = ( + self._xml_tag_buffer[: -len(kept_suffix)] + if kept_suffix + else self._xml_tag_buffer + ) + # Use cached value type for consistency value_type = self._cached_value_type or "string" @@ -425,14 +446,12 @@ def _process_xml_to_json_streaming( 1:-1 ] self._current_value += content - self._xml_tag_buffer = "" elif value_type == "number": if content: if not self._value_started: self._value_started = True json_output += content self._current_value += content - self._xml_tag_buffer = "" else: # For object/array types, output as-is if content: @@ -440,7 +459,8 @@ def _process_xml_to_json_streaming( self._value_started = True json_output += content self._current_value += content - self._xml_tag_buffer = "" + + self._xml_tag_buffer = kept_suffix return json_output diff --git a/test/registered/unit/function_call/test_function_call_parser.py b/test/registered/unit/function_call/test_function_call_parser.py index d021694cb7ea..7087b6d298dc 100644 --- a/test/registered/unit/function_call/test_function_call_parser.py +++ b/test/registered/unit/function_call/test_function_call_parser.py @@ -2618,6 +2618,77 @@ def test_whitespace_preserved_in_arg_values(self): self.assertEqual(params["old_string"], " indented code") self.assertEqual(params["new_string"], " also indented") + def test_streaming_value_ending_with_less_than(self): + """Streaming: a value whose last character overlaps the start of . + + When the value ends with '<', the next character emitted by the model + is the '<' that opens the closing tag. Without correct prefix-suffix + sliding, the buffer becomes '<<', fails the prefix check, and the + entire buffer is ejected as content — at which point the actual + closing tag is never matched and "" leaks into the + emitted JSON arguments. + """ + + def _stream(chunks): + detector = Glm47MoeDetector() + emitted = "" + for chunk in chunks: + result = detector.parse_streaming_increment(chunk, self.tools) + for call in result.calls: + if call.parameters: + emitted += call.parameters + return emitted + + # Single chunk, value ending in '<'. + text = ( + "get_weather" + "city.<" + "date2024-06-27" + "" + ) + emitted = _stream([text]) + self.assertEqual(json.loads(emitted), {"city": ".<", "date": "2024-06-27"}) + + # Realistic chunking: value and closing tag arrive in separate chunks. + emitted = _stream( + [ + "get_weathercity", + ".<", + "date" + "2024-06-27", + ] + ) + self.assertEqual(json.loads(emitted), {"city": ".<", "date": "2024-06-27"}) + + # Pathological chunking: every character on its own. + emitted = _stream(list(text)) + self.assertEqual(json.loads(emitted), {"city": ".<", "date": "2024-06-27"}) + + def test_streaming_value_ending_with_closing_tag_prefix(self): + """Streaming: a value ending with a longer prefix of . + + Same class of bug as the trailing-'<' case, but with a multi-char + overlap. The state machine must drop the smallest leading chunk such + that what remains is a prefix of "" — not eject the whole + buffer. + """ + detector = Glm47MoeDetector() + text = ( + "get_weather" + "city" + "date2024-06-27" + "" + ) + emitted = "" + for chunk in list(text): + result = detector.parse_streaming_increment(chunk, self.tools) + for call in result.calls: + if call.parameters: + emitted += call.parameters + self.assertEqual( + json.loads(emitted), {"city": "