diff --git a/python/sglang/srt/function_call/glm47_moe_detector.py b/python/sglang/srt/function_call/glm47_moe_detector.py
index 9bc11d7703e8..05b9c5413625 100644
--- a/python/sglang/srt/function_call/glm47_moe_detector.py
+++ b/python/sglang/srt/function_call/glm47_moe_detector.py
@@ -412,7 +412,28 @@ def _process_xml_to_json_streaming(
) and closing_tag.startswith(self._xml_tag_buffer)
if not is_potential_closing:
- content = self._xml_tag_buffer
+ # The buffer is not a prefix of "", but a
+ # proper suffix of it might still be. Slide forward to
+ # the smallest k > 0 such that buffer[k:] is a prefix
+ # of the closing tag and keep that suffix buffered;
+ # everything before it is real value content. Without
+ # this, a value ending with "<" (e.g. ".<") followed
+ # immediately by "" would yield buffer
+ # "<<", get ejected wholesale, and the actual closing
+ # tag would never match — leaking "" into
+ # the emitted JSON arguments.
+ kept_suffix = ""
+ for k in range(1, len(self._xml_tag_buffer)):
+ candidate = self._xml_tag_buffer[k:]
+ if closing_tag.startswith(candidate):
+ kept_suffix = candidate
+ break
+ content = (
+ self._xml_tag_buffer[: -len(kept_suffix)]
+ if kept_suffix
+ else self._xml_tag_buffer
+ )
+
# Use cached value type for consistency
value_type = self._cached_value_type or "string"
@@ -425,14 +446,12 @@ def _process_xml_to_json_streaming(
1:-1
]
self._current_value += content
- self._xml_tag_buffer = ""
elif value_type == "number":
if content:
if not self._value_started:
self._value_started = True
json_output += content
self._current_value += content
- self._xml_tag_buffer = ""
else:
# For object/array types, output as-is
if content:
@@ -440,7 +459,8 @@ def _process_xml_to_json_streaming(
self._value_started = True
json_output += content
self._current_value += content
- self._xml_tag_buffer = ""
+
+ self._xml_tag_buffer = kept_suffix
return json_output
diff --git a/test/registered/unit/function_call/test_function_call_parser.py b/test/registered/unit/function_call/test_function_call_parser.py
index d021694cb7ea..7087b6d298dc 100644
--- a/test/registered/unit/function_call/test_function_call_parser.py
+++ b/test/registered/unit/function_call/test_function_call_parser.py
@@ -2618,6 +2618,77 @@ def test_whitespace_preserved_in_arg_values(self):
self.assertEqual(params["old_string"], " indented code")
self.assertEqual(params["new_string"], " also indented")
+ def test_streaming_value_ending_with_less_than(self):
+ """Streaming: a value whose last character overlaps the start of .
+
+ When the value ends with '<', the next character emitted by the model
+ is the '<' that opens the closing tag. Without correct prefix-suffix
+ sliding, the buffer becomes '<<', fails the prefix check, and the
+ entire buffer is ejected as content — at which point the actual
+ closing tag is never matched and "" leaks into the
+ emitted JSON arguments.
+ """
+
+ def _stream(chunks):
+ detector = Glm47MoeDetector()
+ emitted = ""
+ for chunk in chunks:
+ result = detector.parse_streaming_increment(chunk, self.tools)
+ for call in result.calls:
+ if call.parameters:
+ emitted += call.parameters
+ return emitted
+
+ # Single chunk, value ending in '<'.
+ text = (
+ "get_weather"
+ "city.<"
+ "date2024-06-27"
+ ""
+ )
+ emitted = _stream([text])
+ self.assertEqual(json.loads(emitted), {"city": ".<", "date": "2024-06-27"})
+
+ # Realistic chunking: value and closing tag arrive in separate chunks.
+ emitted = _stream(
+ [
+ "get_weathercity",
+ ".<",
+ "date"
+ "2024-06-27",
+ ]
+ )
+ self.assertEqual(json.loads(emitted), {"city": ".<", "date": "2024-06-27"})
+
+ # Pathological chunking: every character on its own.
+ emitted = _stream(list(text))
+ self.assertEqual(json.loads(emitted), {"city": ".<", "date": "2024-06-27"})
+
+ def test_streaming_value_ending_with_closing_tag_prefix(self):
+ """Streaming: a value ending with a longer prefix of .
+
+ Same class of bug as the trailing-'<' case, but with a multi-char
+ overlap. The state machine must drop the smallest leading chunk such
+ that what remains is a prefix of "" — not eject the whole
+ buffer.
+ """
+ detector = Glm47MoeDetector()
+ text = (
+ "get_weather"
+ "city"
+ "date2024-06-27"
+ ""
+ )
+ emitted = ""
+ for chunk in list(text):
+ result = detector.parse_streaming_increment(chunk, self.tools)
+ for call in result.calls:
+ if call.parameters:
+ emitted += call.parameters
+ self.assertEqual(
+ json.loads(emitted), {"city": "