From 6ac4e6b36cb3e3fb954c64f2575cc72474154fc6 Mon Sep 17 00:00:00 2001 From: Bugen Zhao Date: Wed, 2 Sep 2026 20:10:13 +0000 Subject: [PATCH] [Perf][Rust Frontend] Coalesce decoded chunks per engine update Assisted-by: OpenAI Codex Signed-off-by: Bugen Zhao --- rust/src/text/src/output/decoded.rs | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/rust/src/text/src/output/decoded.rs b/rust/src/text/src/output/decoded.rs index 0dec802f5b91..a438c8e267d6 100644 --- a/rust/src/text/src/output/decoded.rs +++ b/rust/src/text/src/output/decoded.rs @@ -201,11 +201,14 @@ pub async fn decoded_text_event_stream( break; } + } - // TODO: avoid generating a chunk every time a token is pushed - if intermediate && let Some(chunk) = decoder.next_chunk() { - decoded.append(chunk); - } + // Coalesce output per nonterminal engine update; terminal updates flush below. + if intermediate + && finish_reason.is_none() + && let Some(chunk) = decoder.next_chunk() + { + decoded.append(chunk); } let mut new_token_ids = output.token_ids;