From 439422e207e6ff6aa8e9d8607896039696533257 Mon Sep 17 00:00:00 2001 From: Kan Wu Date: Fri, 25 Sep 2026 06:39:22 +0000 Subject: [PATCH 1/5] [sgl-router] Forward input_ids for every non-multimodal chat Drop the per-shape forwarding guard; only multimodal chats and caller-provided input_ids are excluded. The native DeepSeek-V4 renderer is fixture-verified for these shapes (new cases: mid-conversation system, final assistant turn, agentic thinking history, tool content parts, reasoning object, ignored fields; SGLang main and v0.5.20 agree). Other formatters are unverified, so the router logs a loud warning at startup. Co-Authored-By: Claude Opus 5.5 --- .../src/server/routes/chat/preparation.rs | 265 +----------------- .../src/tokenizer/chat_formatter.rs | 5 + experimental/sgl-router/src/tokenizer/mod.rs | 22 +- .../tests/fixtures/deepseek/v4.json | 9 +- .../tests/proxy/cache_aware_input_ids.rs | 47 ---- .../tests/proxy/roundrobin_input_ids.rs | 112 +------- .../tests/proxy/sticky_input_ids.rs | 51 +--- .../tests/scripts/generate_deepseek_parity.py | 2 +- 8 files changed, 43 insertions(+), 470 deletions(-) diff --git a/experimental/sgl-router/src/server/routes/chat/preparation.rs b/experimental/sgl-router/src/server/routes/chat/preparation.rs index 3a2ce8dcefeb..7760b21a8c1d 100644 --- a/experimental/sgl-router/src/server/routes/chat/preparation.rs +++ b/experimental/sgl-router/src/server/routes/chat/preparation.rs @@ -516,58 +516,10 @@ fn build_outgoing_body( Ok(Bytes::from(bytes)) } -/// Forward generated IDs only for request shapes verified against the engine. -/// The engine uses `input_ids` verbatim, bypassing its chat-template processing. -/// -/// Preserve caller-provided IDs. Exclude requests that may render differently -/// with dynamo-render: -/// - Non-leading system turns or consecutive users, which strict templates rewrite. -/// - Historical `reasoning_content`, which may be injected into message content. -/// - Tools and tool-call history, which the engine merges and normalizes -/// before rendering. -/// - Non-string or missing content, which the engine flattens or blanks. -/// - Template overrides, kwargs, reasoning controls, or task selection. -/// - Assistant continuations, whose final turn the engine handles separately. -/// -/// Matching model files and engine defaults are still required. Worker template -/// overrides and default kwargs cannot be inferred from the request. -/// `--disable-input-ids-forwarding` gates forwarding separately for such fleets. -fn can_forward_chat_tokens(value: &Value) -> bool { - if has_caller_input_ids(value) - || request_has_tools(value) - || request_has_non_text_content(value) - || request_has_reasoning_content(value) - || request_has_role_rewrites(value) - { - return false; - } - // Request controls whose rendering has not been verified against the engine. - for key in [ - "chat_template", - "chat_template_kwargs", - "reasoning", - "reasoning_effort", - "task", - ] { - if value.get(key).is_some_and(|v| !v.is_null()) { - return false; - } - } - if value - .get("continue_final_message") - .and_then(|v| v.as_bool()) - == Some(true) - { - return false; - } - !last_message_is_assistant(value) -} - /// Whether router-rendered `input_ids` replace engine tokenization. /// -/// Only chats with forwarding enabled that pass the forwarding guard are -/// eligible; an eligible chat without chat-rendered tokens is a failed offload. -/// Multimodal chats are reported apart from other guard exclusions. +/// Every text chat is eligible; multimodal chats and caller-provided IDs are not. +/// An eligible chat without chat-rendered tokens is a failed offload. fn input_ids_forwarding( can_forward_input_ids: bool, request_value: Option<&Value>, @@ -580,7 +532,7 @@ fn input_ids_forwarding( return InputIdsForwarding::IneligibleMultimodal; } let eligible = request_value.is_some_and(|v| { - v.get("messages").is_some_and(|m| m.is_array()) && can_forward_chat_tokens(v) + v.get("messages").is_some_and(|m| m.is_array()) && !has_caller_input_ids(v) }); if !eligible { InputIdsForwarding::Ineligible @@ -591,86 +543,8 @@ fn input_ids_forwarding( } } -/// Whether the final chat message has `role: "assistant"` (a prefix / -/// continuation turn the engine's template path special-cases). -fn last_message_is_assistant(value: &Value) -> bool { - value - .get("messages") - .and_then(|m| m.as_array()) - .and_then(|msgs| msgs.last()) - .and_then(|m| m.get("role")) - .and_then(|r| r.as_str()) - == Some("assistant") -} - -/// Tool schemas and tool-call history require engine normalization before -/// rendering: the engine merges message-level `tools` into the template's tools -/// and parses `tool_calls` arguments; dynamo-render does neither the same way. -fn request_has_tools(value: &Value) -> bool { - let nonempty = |v: &Value| match v { - Value::Array(a) => !a.is_empty(), - Value::Null => false, - _ => true, - }; - if ["tools", "functions"] - .iter() - .any(|key| value.get(key).is_some_and(nonempty)) - { - return true; - } - value - .get("messages") - .and_then(|m| m.as_array()) - .is_some_and(|messages| { - messages.iter().any(|message| { - message["role"] == "tool" - || ["tools", "tool_calls", "function_call"] - .iter() - .any(|key| message.get(key).is_some_and(nonempty)) - }) - }) -} - -/// dynamo-render may inject historical reasoning into content the engine leaves unchanged. -fn request_has_reasoning_content(value: &Value) -> bool { - value - .get("messages") - .and_then(|messages| messages.as_array()) - .is_some_and(|messages| { - messages.iter().any(|message| { - message - .get("reasoning_content") - .is_some_and(|v| !v.is_null()) - }) - }) -} - -/// Message orders dynamo-render may rewrite for strict templates. -fn request_has_role_rewrites(value: &Value) -> bool { - let Some(messages) = value.get("messages").and_then(|v| v.as_array()) else { - return false; - }; - messages.iter().skip(1).any(|m| m["role"] == "system") - || messages - .windows(2) - .any(|pair| pair[0]["role"] == "user" && pair[1]["role"] == "user") -} - -/// Detect non-string or missing content, which requires engine tokenization: -/// the engine normalizes arrays and nulls differently from dynamo-render. -fn request_has_non_text_content(value: &Value) -> bool { - value - .get("messages") - .and_then(|m| m.as_array()) - .is_some_and(|msgs| { - msgs.iter() - .any(|m| !matches!(m.get("content"), Some(Value::String(_)))) - }) -} - /// Image, video, and audio content parts need the engine's multimodal -/// processor, so such chats can never carry `input_ids`. Other non-string -/// parts (`refusal`, `thinking`, untyped) are ordinary guard exclusions. +/// processor, so such chats can never carry `input_ids`. fn request_has_multimodal_content(value: &Value) -> bool { value .get("messages") @@ -840,42 +714,6 @@ mod tests { } } - #[test] - fn request_has_tools_detects_tools_and_functions() { - assert!(request_has_tools(&json!({"tools":[{"type":"function"}]}))); - assert!(request_has_tools(&json!({"functions":[{"name":"f"}]}))); - assert!(!request_has_tools(&json!({"tools":[]}))); - assert!(!request_has_tools(&json!({"messages":[]}))); - for message in [ - json!({"role":"system","content":"s","tools":[{"type":"function"}]}), - json!({"role":"assistant","content":"","tool_calls":[{"function":{"name":"f","arguments":"{}"}}]}), - ] { - assert!(request_has_tools(&json!({"messages":[message]}))); - } - } - - #[test] - fn request_has_non_text_content_detects_non_string_content() { - for content in [ - json!([{"type":"image_url","image_url":"x"}]), - json!([{"type":"text","text":"a"},{"type":"text","text":"b"}]), - Value::Null, - ] { - assert!( - request_has_non_text_content(&json!({ - "messages":[{"role":"user","content":"hi"},{"role":"assistant","content":content}] - })), - "content {content} must block" - ); - } - assert!(request_has_non_text_content(&json!({ - "messages":[{"role":"assistant","tool_calls":[]}] - }))); - assert!(!request_has_non_text_content(&json!({ - "messages":[{"role":"user","content":"hello"}] - }))); - } - #[test] fn request_has_multimodal_content_detects_media_parts() { for part in [ @@ -909,95 +747,6 @@ mod tests { } } - #[test] - fn reasoning_history_is_an_expected_forwarding_omission() { - let mut value = json!({"messages": [ - {"role":"user", "content":"hi"}, - {"role":"assistant", "content":"answer", "reasoning_content":"prior reasoning"}, - {"role":"user", "content":"next"} - ]}); - assert!(!can_forward_chat_tokens(&value)); - assert_eq!( - input_ids_forwarding(true, Some(&value), None), - InputIdsForwarding::Ineligible - ); - value["messages"][1]["reasoning_content"] = Value::Null; - assert!(can_forward_chat_tokens(&value)); - value["messages"][1] - .as_object_mut() - .unwrap() - .remove("reasoning_content"); - assert!(can_forward_chat_tokens(&value)); - } - - #[test] - fn role_rewrites_are_expected_forwarding_omissions() { - for roles in [ - vec!["user", "user"], - vec!["system", "system", "user"], - vec!["user", "assistant", "system", "user"], - ] { - let messages: Vec<_> = roles - .iter() - .map(|role| json!({"role": role, "content": "text"})) - .collect(); - let value = json!({"messages": messages}); - assert!(!can_forward_chat_tokens(&value), "{roles:?}"); - assert_eq!( - input_ids_forwarding(true, Some(&value), None), - InputIdsForwarding::Ineligible - ); - } - assert!(can_forward_chat_tokens(&json!({"messages": [ - {"role": "system", "content": "instructions"}, - {"role": "user", "content": "hi"}, - {"role": "assistant", "content": "hello"}, - {"role": "user", "content": "next"} - ]}))); - } - - #[test] - fn can_forward_chat_tokens_allows_plain_text_chat() { - assert!(can_forward_chat_tokens(&json!({ - "messages": [{"role": "user", "content": "hello"}] - }))); - } - - #[test] - fn can_forward_chat_tokens_blocks_unreplicated_signals() { - let blockers = [ - json!({"messages":[{"role":"user","content":"hi"}],"input_ids":[7, 8]}), - json!({"messages":[{"role":"user","content":"hi"}],"input_ids":"bad"}), - json!({"messages":[{"role":"user","content":"hi"}],"tools":[{"type":"function"}]}), - json!({"messages":[{"role":"user","content":[{"type":"image_url","image_url":"x"}]}]}), - json!({"messages":[{"role":"user","content":"hi"}],"chat_template":"{{ custom }}"}), - json!({"messages":[{"role":"user","content":"hi"}],"chat_template_kwargs":{"enable_thinking":true}}), - json!({"messages":[{"role":"user","content":"hi"}],"reasoning_effort":"high"}), - json!({"messages":[{"role":"user","content":"hi"}],"reasoning":{"enabled":true}}), - json!({"messages":[{"role":"user","content":"hi"}],"task":"generate"}), - json!({"messages":[{"role":"user","content":"hi"}],"continue_final_message":true}), - json!({"messages":[{"role":"user","content":"hi"},{"role":"assistant","content":"partial"}]}), - ]; - for b in blockers { - assert!( - !can_forward_chat_tokens(&b), - "must NOT forward input_ids for: {b}" - ); - } - } - - #[test] - fn can_forward_chat_tokens_ignores_null_and_false_fields() { - assert!(can_forward_chat_tokens(&json!({ - "messages": [{"role": "user", "content": "hi"}], - "input_ids": null, - "chat_template": null, - "reasoning_effort": null, - "chat_template_kwargs": null, - "continue_final_message": false - }))); - } - #[test] fn input_ids_forwarding_outcome_per_request() { use InputIdsForwarding::*; @@ -1010,17 +759,19 @@ mod tests { ]}]}); let text_parts = json!({"messages":[{"role":"user","content":[{"type":"text","text":"hi"}]}]}); + let caller_ids = json!({"messages":[{"role":"user","content":"hi"}], "input_ids":[7]}); for (enabled, value, rendered, expected) in [ (true, Some(&chat), Some(true), Forwarded), (true, Some(&chat), Some(false), TokenizeFailed), (true, Some(&chat), None, TokenizeFailed), (false, Some(&chat), Some(true), Disabled), - (true, Some(&tools), Some(true), Ineligible), + (true, Some(&tools), Some(true), Forwarded), + (true, Some(&caller_ids), Some(false), Ineligible), (true, Some(&prompt), None, Ineligible), (true, None, None, Ineligible), (true, Some(&image), None, IneligibleMultimodal), (false, Some(&image), None, Disabled), - (true, Some(&text_parts), None, Ineligible), + (true, Some(&text_parts), Some(true), Forwarded), ] { let tokens = rendered.map(|rendered_from_chat| RequestTokens { ids: vec![1, 2, 3], diff --git a/experimental/sgl-router/src/tokenizer/chat_formatter.rs b/experimental/sgl-router/src/tokenizer/chat_formatter.rs index bd630affe344..2c786c8f4c6a 100644 --- a/experimental/sgl-router/src/tokenizer/chat_formatter.rs +++ b/experimental/sgl-router/src/tokenizer/chat_formatter.rs @@ -231,6 +231,11 @@ impl ChatFormatter { }) } + /// Whether router rendering is fixture-verified against SGLang for every text chat. + pub fn forwarding_verified(&self) -> bool { + matches!(self.deepseek, Some(super::deepseek::Encoder::V4(_))) + } + /// Apply the workers' `--default-chat-template-kwargs`; they fill keys the /// request leaves unset, and a default `reasoning_effort` acts as the request's. pub fn with_defaults(mut self, defaults: &ChatTemplateKwargs) -> Self { diff --git a/experimental/sgl-router/src/tokenizer/mod.rs b/experimental/sgl-router/src/tokenizer/mod.rs index 1f4336cb9b9d..f607b61dcbfd 100644 --- a/experimental/sgl-router/src/tokenizer/mod.rs +++ b/experimental/sgl-router/src/tokenizer/mod.rs @@ -82,14 +82,20 @@ impl TokenizerRegistry { tracing::info!(model = %m.id, "router-generated input_ids forwarding disabled; workers tokenize messages; \ routing tokenization remains available"); - } else if me.has_chat_formatter(&m.id) { - tracing::warn!(model = %m.id, - "router-generated input_ids forwarding enabled: requires the workers' model files, \ - --default-chat-template-kwargs, SGLANG_DEFAULT_THINKING, and \ - SGLANG_DSV4_REASONING_EFFORT / SGLANG_DSV41_REASONING_EFFORT; worker parser overrides \ - (including --tool-call-parser deepseekv32), content-format detection, and \ - conversation-template stop strings are not replicated. Use \ - --disable-input-ids-forwarding for array-only templates or when these assumptions do not hold"); + } else if let Some(entry) = me.formatters.get(&m.id) { + if entry.formatter.forwarding_verified() { + tracing::info!(model = %m.id, + "router-generated input_ids forwarding enabled for all text chats; requires the \ + workers' model files, --default-chat-template-kwargs, SGLANG_DEFAULT_THINKING, \ + and SGLANG_DSV4_REASONING_EFFORT"); + } else { + tracing::warn!(model = %m.id, + "UNVERIFIED input_ids forwarding: router rendering is verified against SGLang \ + only for DeepSeek-V4; this model's chats may be forwarded with prompts that \ + differ from what the workers would render (tools, reasoning history, content \ + parts, strict templates). Pass --disable-input-ids-forwarding unless you have \ + verified parity for this model"); + } } Ok(me) } diff --git a/experimental/sgl-router/tests/fixtures/deepseek/v4.json b/experimental/sgl-router/tests/fixtures/deepseek/v4.json index 1582952768f6..c6172e5f1434 100644 --- a/experimental/sgl-router/tests/fixtures/deepseek/v4.json +++ b/experimental/sgl-router/tests/fixtures/deepseek/v4.json @@ -22,5 +22,12 @@ {"name":"preview/effort_max","profile":"preview","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"max"},"prompt":"<|begin▁of▁sentence|>Reasoning Effort: Absolute maximum with no shortcuts permitted.\nYou MUST be very thorough in your thinking and comprehensively decompose the problem to resolve the root cause, rigorously stress-testing your logic against all potential paths, edge cases, and adversarial scenarios.\nExplicitly write out your entire deliberation process, documenting every intermediate step, considered alternative, and rejected hypothesis to ensure absolutely no assumption is left unchecked.\n\n<|User|>Hi<|Assistant|>","token_count":84,"token_sha256":"daf853d041e8dde87f07e95a201f8f07459c88482fe53e46472605d63fbdbeac"}, {"name":"preview/task_action","profile":"preview","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"task":"action"},"prompt":"<|begin▁of▁sentence|><|User|>Hi<|Assistant|><|action|>","token_count":6,"token_sha256":"925bcaae248cc41efabd3e5d8f660f17513e43174f1d34aba51a1399aba76e8e"}, {"name":"official/effort_high","profile":"official","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"high"},"prompt":"<|begin▁of▁sentence|>Reasoning Effort: Absolute maximum with no shortcuts permitted.\nYou MUST be very thorough in your thinking and comprehensively decompose the problem to resolve the root cause, rigorously stress-testing your logic against all potential paths, edge cases, and adversarial scenarios.\nExplicitly write out your entire deliberation process, documenting every intermediate step, considered alternative, and rejected hypothesis to ensure absolutely no assumption is left unchecked.\n\n<|User|>Hi<|Assistant|>","token_count":84,"token_sha256":"daf853d041e8dde87f07e95a201f8f07459c88482fe53e46472605d63fbdbeac"}, -{"name":"official/effort_max","profile":"official","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"max"},"prompt":"<|begin▁of▁sentence|>Reasoning Effort: Beyond maximum — exhaustive, relentless, and uncompromising.\nYou MUST reason with the utmost depth and rigor, leaving absolutely nothing to chance: exhaustively decompose the problem into its most fundamental components, trace every causal chain to its root, and resolve the underlying cause rather than any surface symptom.\nDo not stop reasoning until you have independently verified the solution from multiple angles and are certain that no assumption remains unchecked and no error remains undiscovered.\n\n<|User|>Hi<|Assistant|>","token_count":97,"token_sha256":"82ee92542b8545282dcc47a141cb9262d6ab3e20463e0baf445557ca0a9b905c"} +{"name":"official/effort_max","profile":"official","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"max"},"prompt":"<|begin▁of▁sentence|>Reasoning Effort: Beyond maximum — exhaustive, relentless, and uncompromising.\nYou MUST reason with the utmost depth and rigor, leaving absolutely nothing to chance: exhaustively decompose the problem into its most fundamental components, trace every causal chain to its root, and resolve the underlying cause rather than any surface symptom.\nDo not stop reasoning until you have independently verified the solution from multiple angles and are certain that no assumption remains unchecked and no error remains undiscovered.\n\n<|User|>Hi<|Assistant|>","token_count":97,"token_sha256":"82ee92542b8545282dcc47a141cb9262d6ab3e20463e0baf445557ca0a9b905c"}, +{"name":"official/mid_system","profile":"official","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"system","content":"Now be terse"},{"role":"user","content":"Again"}]},"prompt":"<|begin▁of▁sentence|><|User|>HiNow be terse<|User|>Again<|Assistant|>","token_count":11,"token_sha256":"a7b0adcf282b572be34f0c5c5cf1697c84a04e8f7c00aeb55e2e23659d8000f5"}, +{"name":"official/final_assistant","profile":"official","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"assistant","content":"Partial"}]},"prompt":"<|begin▁of▁sentence|><|User|>Hi\n\nPartial<|Assistant|>","token_count":7,"token_sha256":"f5287f4be7bd7e716fbc134e7f2d54de845a8998ab4d5e2e45ffbd359ee0269a"}, +{"name":"official/agentic_thinking","profile":"official","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"assistant","content":null,"reasoning_content":"Need foo","tool_calls":[{"id":"a","type":"function","function":{"name":"foo","arguments":"{\"x\":\"1\"}"}}]},{"role":"tool","tool_call_id":"a","content":"R1"},{"role":"assistant","content":"Done","reasoning_content":"Got it"},{"role":"user","content":"Next"}],"tools":[{"type":"function","function":{"name":"foo","description":"Foo","parameters":{"type":"object","properties":{"x":{"type":"string"}}}}}],"chat_template_kwargs":{"thinking":true},"reasoning_effort":"high"},"prompt":"<|begin▁of▁sentence|>Reasoning Effort: Absolute maximum with no shortcuts permitted.\nYou MUST be very thorough in your thinking and comprehensively decompose the problem to resolve the root cause, rigorously stress-testing your logic against all potential paths, edge cases, and adversarial scenarios.\nExplicitly write out your entire deliberation process, documenting every intermediate step, considered alternative, and rejected hypothesis to ensure absolutely no assumption is left unchecked.\n\n\n\n## Tools\n\nYou have access to a set of tools to help answer the user's question. You can invoke tools by writing a \"<|DSML|tool_calls>\" block like the following:\n\n<|DSML|tool_calls>\n<|DSML|invoke name=\"$TOOL_NAME\">\n<|DSML|parameter name=\"$PARAMETER_NAME\" string=\"true|false\">$PARAMETER_VALUE\n...\n\n<|DSML|invoke name=\"$TOOL_NAME2\">\n...\n\n\n\nString parameters should be specified as is and set `string=\"true\"`. For all other types (numbers, booleans, arrays, objects), pass the value in JSON format and set `string=\"false\"`.\n\nIf thinking_mode is enabled (triggered by ), you MUST output your complete reasoning inside ... BEFORE any tool calls or final response.\n\nOtherwise, output directly after with tool calls or final response.\n\n### Available Tool Schemas\n\n{\"description\": \"Foo\", \"name\": \"foo\", \"parameters\": {\"type\": \"object\", \"properties\": {\"x\": {\"type\": \"string\"}}}, \"strict\": false}\n\nYou MUST strictly follow the above defined tool name and parameter schemas to invoke tool calls.\n<|User|>Hi<|Assistant|>Need foo\n\n<|DSML|tool_calls>\n<|DSML|invoke name=\"foo\">\n<|DSML|parameter name=\"x\" string=\"true\">1\n\n<|end▁of▁sentence|><|User|>R1<|Assistant|>Got itDone<|end▁of▁sentence|><|User|>Next<|Assistant|>","token_count":413,"token_sha256":"861c4197c0329945e94d8051233e7a7c4d619ea252430bfa3a418df1269aec45"}, +{"name":"official/agentic_thinking_open","profile":"official","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"assistant","content":"","reasoning_content":"Need foo","tool_calls":[{"id":"a","type":"function","function":{"name":"foo","arguments":"{\"x\":\"1\"}"}},{"id":"b","type":"function","function":{"name":"foo","arguments":"{\"x\":\"2\"}"}}]},{"role":"tool","tool_call_id":"a","content":[{"type":"text","text":"R1"}]},{"role":"tool","tool_call_id":"b","content":"R2"}],"tools":[{"type":"function","function":{"name":"foo","description":"Foo","parameters":{"type":"object","properties":{"x":{"type":"string"}}}}}],"chat_template_kwargs":{"thinking":true}},"prompt":"<|begin▁of▁sentence|>\n\n## Tools\n\nYou have access to a set of tools to help answer the user's question. You can invoke tools by writing a \"<|DSML|tool_calls>\" block like the following:\n\n<|DSML|tool_calls>\n<|DSML|invoke name=\"$TOOL_NAME\">\n<|DSML|parameter name=\"$PARAMETER_NAME\" string=\"true|false\">$PARAMETER_VALUE\n...\n\n<|DSML|invoke name=\"$TOOL_NAME2\">\n...\n\n\n\nString parameters should be specified as is and set `string=\"true\"`. For all other types (numbers, booleans, arrays, objects), pass the value in JSON format and set `string=\"false\"`.\n\nIf thinking_mode is enabled (triggered by ), you MUST output your complete reasoning inside ... BEFORE any tool calls or final response.\n\nOtherwise, output directly after with tool calls or final response.\n\n### Available Tool Schemas\n\n{\"description\": \"Foo\", \"name\": \"foo\", \"parameters\": {\"type\": \"object\", \"properties\": {\"x\": {\"type\": \"string\"}}}, \"strict\": false}\n\nYou MUST strictly follow the above defined tool name and parameter schemas to invoke tool calls.\n<|User|>Hi<|Assistant|>Need foo\n\n<|DSML|tool_calls>\n<|DSML|invoke name=\"foo\">\n<|DSML|parameter name=\"x\" string=\"true\">1\n\n<|DSML|invoke name=\"foo\">\n<|DSML|parameter name=\"x\" string=\"true\">2\n\n<|end▁of▁sentence|><|User|>R1\n\nR2<|Assistant|>","token_count":365,"token_sha256":"950bf1008e84d681ad49fad2d11fd038c14b1d16d75136da19b7ec94b9be94bb"}, +{"name":"official/thinking_multi_turn","profile":"official","request":{"model":"m","messages":[{"role":"system","content":"S"},{"role":"user","content":"Hi"},{"role":"assistant","content":"A1","reasoning_content":"R1"},{"role":"user","content":"Q2"},{"role":"assistant","content":"A2","reasoning_content":"R2"},{"role":"user","content":"Q3"}],"chat_template_kwargs":{"thinking":true}},"prompt":"<|begin▁of▁sentence|>S<|User|>Hi<|Assistant|>A1<|end▁of▁sentence|><|User|>Q2<|Assistant|>A2<|end▁of▁sentence|><|User|>Q3<|Assistant|>","token_count":22,"token_sha256":"ea52887320284b65ab2dd2d732ac66b7a1cd111e7d0d40a78fe8f6c0cc05c0df"}, +{"name":"official/reasoning_object","profile":"official","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning":{"effort":"max"}},"prompt":"<|begin▁of▁sentence|>Reasoning Effort: Beyond maximum — exhaustive, relentless, and uncompromising.\nYou MUST reason with the utmost depth and rigor, leaving absolutely nothing to chance: exhaustively decompose the problem into its most fundamental components, trace every causal chain to its root, and resolve the underlying cause rather than any surface symptom.\nDo not stop reasoning until you have independently verified the solution from multiple angles and are certain that no assumption remains unchecked and no error remains undiscovered.\n\n<|User|>Hi<|Assistant|>","token_count":97,"token_sha256":"82ee92542b8545282dcc47a141cb9262d6ab3e20463e0baf445557ca0a9b905c"}, +{"name":"official/ignored_fields","profile":"official","request":{"model":"m","messages":[{"role":"user","content":"Hi","name":"kan"}],"response_format":{"type":"json_object"},"stream":true,"max_tokens":8},"prompt":"<|begin▁of▁sentence|><|User|>Hi<|Assistant|>","token_count":5,"token_sha256":"95aeed952d0173ec8849b13faec66eb61ec4412cd9c41dae0c405257559365a6"} ]} diff --git a/experimental/sgl-router/tests/proxy/cache_aware_input_ids.rs b/experimental/sgl-router/tests/proxy/cache_aware_input_ids.rs index 1131c1863456..959cad51f21e 100644 --- a/experimental/sgl-router/tests/proxy/cache_aware_input_ids.rs +++ b/experimental/sgl-router/tests/proxy/cache_aware_input_ids.rs @@ -101,53 +101,6 @@ async fn plain_chat_forwards_input_ids_and_keeps_messages() { ); } -#[tokio::test] -async fn tool_request_omits_input_ids() { - let mock = MockWorker::start(vec![]).await; - let ctx = build_ctx(mock.url.clone()); - let status = send( - ctx, - json!({ - "model": MODEL, - "messages": [{"role": "user", "content": "hi"}], - "tools": [{"type": "function", "function": {"name": "f"}}], - }), - ) - .await; - assert_eq!(status, StatusCode::OK); - - let body = captured(&mock); - assert!( - body.get("input_ids").is_none(), - "tool requests must not forward input_ids; got {body}" - ); -} - -#[tokio::test] -async fn thinking_request_omits_input_ids() { - // `chat_template_kwargs` steers engine-side thinking mode, which the - // router's encoder renders in the default mode only — forwarding ids would - // silently run the wrong mode, so the handler must omit them. - let mock = MockWorker::start(vec![]).await; - let ctx = build_ctx(mock.url.clone()); - let status = send( - ctx, - json!({ - "model": MODEL, - "messages": [{"role": "user", "content": "hi"}], - "chat_template_kwargs": {"enable_thinking": true}, - }), - ) - .await; - assert_eq!(status, StatusCode::OK); - - let body = captured(&mock); - assert!( - body.get("input_ids").is_none(), - "thinking-mode requests must not forward input_ids; got {body}" - ); -} - #[tokio::test] async fn multimodal_request_omits_input_ids() { let mock = MockWorker::start(vec![]).await; diff --git a/experimental/sgl-router/tests/proxy/roundrobin_input_ids.rs b/experimental/sgl-router/tests/proxy/roundrobin_input_ids.rs index 497d7565c724..634d885bfd94 100644 --- a/experimental/sgl-router/tests/proxy/roundrobin_input_ids.rs +++ b/experimental/sgl-router/tests/proxy/roundrobin_input_ids.rs @@ -272,30 +272,6 @@ async fn disabled_forwarding_does_not_count_routing_render_failures_as_offload_e assert_forwarded_unchanged(&ctx, &mock, &request).await; } -/// Even under round-robin, a tool request omits `input_ids` (the safe predicate -/// is policy-independent too). -#[tokio::test] -async fn round_robin_tool_request_omits_input_ids() { - let mock = MockWorker::start(vec![]).await; - let ctx = build_ctx(mock.url.clone()); - let status = send( - ctx, - json!({ - "model": MODEL, - "messages": [{"role": "user", "content": "hi"}], - "tools": [{"type": "function", "function": {"name": "f"}}], - }), - ) - .await; - assert_eq!(status, StatusCode::OK); - - let body = captured(&mock); - assert!( - body.get("input_ids").is_none(), - "tool requests must not forward input_ids under any policy; got {body}" - ); -} - /// One forwarding outcome books per dispatched chat request. #[tokio::test] async fn input_ids_forwarding_metric_books_outcome_per_request() { @@ -305,9 +281,12 @@ async fn input_ids_forwarding_metric_books_outcome_per_request() { let mut image = chat.clone(); image["messages"][0]["content"] = json!([{"type": "image_url", "image_url": {"url": "data:image/png;base64,AA=="}}]); + let mut caller_ids = chat.clone(); + caller_ids["input_ids"] = json!([1, 2]); for (cfg, request, outcome) in [ (config(), &chat, "forwarded"), - (config(), &tools, "ineligible"), + (config(), &tools, "forwarded"), + (config(), &caller_ids, "ineligible"), (config(), &image, "ineligible_multimodal"), ( without_forwarding(config(), PolicyKind::RoundRobin), @@ -334,10 +313,7 @@ async fn input_ids_forwarding_metric_books_outcome_per_request() { /// A successful plain-chat forward on a chat-formatter model must NOT emit /// `sgl_router_ingress_tokenize_errors_total` — that counter fires only when the -/// offload was expected but the encoder failed. A tool request on the same model -/// is an *expected* omission (its ids are still engine-equivalent; the -/// safe-predicate withholds forwarding for other reasons), so it must not emit -/// the error counter either. +/// offload was expected but the encoder failed. #[tokio::test] async fn successful_forward_does_not_emit_ingress_tokenize_error() { let mock = MockWorker::start(vec![]).await; @@ -371,84 +347,8 @@ async fn successful_forward_does_not_emit_ingress_tokenize_error() { ); assert!( !m.contains("sgl_router_ingress_tokenize_errors_total{"), - "healthy forwards (and expected omissions) must not emit the error counter; got:\n{m}", - ); -} - -/// History that dynamo-render rewrites stays intact for engine-side tokenization. -#[tokio::test] -async fn reasoning_history_preserves_messages_without_forwarding_ids() { - let (_dir, cfg) = template_config(json!({ - "chat_template": "{% for m in messages %}{{ m.role }}:{{ m.content }};{% endfor %}" - })); - let mock = MockWorker::start(vec![]).await; - let ctx = build_ctx_with_config(mock.url.clone(), cfg); - let mut request = json!({"model": MODEL, "messages": [ - {"role":"user", "content":"hi"}, - {"role":"assistant", "content":"answer", "reasoning_content":"prior reasoning"}, - {"role":"user", "content":"next"} - ]}); - assert!(!ctx - .tokenizers - .encode_chat(MODEL, &request) - .unwrap() - .is_empty()); - assert_forwarded_unchanged(&ctx, &mock, &request).await; - - request["messages"][1] - .as_object_mut() - .unwrap() - .remove("reasoning_content"); - assert_eq!(send(ctx, request).await, StatusCode::OK); - assert!(captured(&mock).get("input_ids").is_some()); -} - -/// Strict-template rewrites are used for routing only; the engine gets the original turns. -#[tokio::test] -async fn role_rewrites_preserve_messages_without_forwarding_ids() { - let template = concat!( - "{%- set ns = namespace(prev='') -%}", - "{%- for m in messages -%}", - "{%- if m.role == 'system' and not loop.first -%}", - "{{ raise_exception('System message must be first.') }}", - "{%- endif -%}", - "{%- if m.role == 'user' and ns.prev == 'user' -%}", - "{{ raise_exception('Conversation roles must alternate.') }}", - "{%- endif -%}", - "{{ m.role }}:{{ m.content }};", - "{%- set ns.prev = m.role -%}", - "{%- endfor -%}" + "healthy forwards must not emit the error counter; got:\n{m}", ); - let (_dir, cfg) = template_config(json!({ - "chat_template": template, "sp_model_kwargs": {"enable_sampling": false} - })); - let mock = MockWorker::start(vec![]).await; - let ctx = build_ctx_with_config(mock.url.clone(), cfg); - for roles in [ - vec!["user", "user"], - vec!["system", "system", "user"], - vec!["user", "assistant", "system", "user"], - ] { - let messages: Vec<_> = roles - .iter() - .map(|role| json!({"role": role, "content": "text"})) - .collect(); - let request = json!({"model": MODEL, "messages": messages}); - assert!(!ctx - .tokenizers - .encode_chat(MODEL, &request) - .unwrap() - .is_empty()); - assert_forwarded_unchanged(&ctx, &mock, &request).await; - } - let request = json!({"model": MODEL, "messages": [ - {"role": "system", "content": "instructions"}, - {"role": "user", "content": "hi"}, - {"role": "assistant", "content": "hello"}, - {"role": "user", "content": "next"} - ]}); - assert_eq!(send(ctx, request).await, StatusCode::OK); - assert!(captured(&mock).get("input_ids").is_some()); } #[path = "../fixtures/kimi_k3.rs"] diff --git a/experimental/sgl-router/tests/proxy/sticky_input_ids.rs b/experimental/sgl-router/tests/proxy/sticky_input_ids.rs index 37187c2d0f36..7763475502f7 100644 --- a/experimental/sgl-router/tests/proxy/sticky_input_ids.rs +++ b/experimental/sgl-router/tests/proxy/sticky_input_ids.rs @@ -11,7 +11,7 @@ //! //! * A plain text chat request forwards `input_ids` AND retains `messages`, //! even though sticky never consults the tokens for routing. -//! * A request carrying `tools` / multimodal content omits `input_ids` — the +//! * A request carrying multimodal content omits `input_ids` — the //! same safe-to-forward predicate applies regardless of policy. //! * Same-session-header requests still pin to a single worker (O(1) sticky //! routing is unchanged by the added tokenization). @@ -160,55 +160,6 @@ async fn sticky_plain_chat_forwards_input_ids_and_keeps_messages() { ); } -#[tokio::test] -async fn sticky_tool_request_omits_input_ids() { - let mock = MockWorker::start(vec![]).await; - let ctx = build_ctx(std::slice::from_ref(&mock.url)); - let status = send( - ctx, - "alice", - json!({ - "model": MODEL, - "messages": [{"role": "user", "content": "hi"}], - "tools": [{"type": "function", "function": {"name": "f"}}], - }), - ) - .await; - assert_eq!(status, StatusCode::OK); - - let body = captured(&mock); - assert!( - body.get("input_ids").is_none(), - "tool requests must not forward input_ids even under sticky; got {body}" - ); -} - -#[tokio::test] -async fn sticky_thinking_request_omits_input_ids() { - // `chat_template_kwargs` steers engine-side thinking mode the router's - // encoder renders in the default mode only — the safe-to-forward predicate - // is policy-independent, so sticky must omit ids here too. - let mock = MockWorker::start(vec![]).await; - let ctx = build_ctx(std::slice::from_ref(&mock.url)); - let status = send( - ctx, - "alice", - json!({ - "model": MODEL, - "messages": [{"role": "user", "content": "hi"}], - "chat_template_kwargs": {"enable_thinking": true}, - }), - ) - .await; - assert_eq!(status, StatusCode::OK); - - let body = captured(&mock); - assert!( - body.get("input_ids").is_none(), - "thinking-mode requests must not forward input_ids under sticky; got {body}" - ); -} - #[tokio::test] async fn sticky_multimodal_request_omits_input_ids() { let mock = MockWorker::start(vec![]).await; diff --git a/experimental/sgl-router/tests/scripts/generate_deepseek_parity.py b/experimental/sgl-router/tests/scripts/generate_deepseek_parity.py index 8d8e8d06f055..6ebf0edc70c5 100644 --- a/experimental/sgl-router/tests/scripts/generate_deepseek_parity.py +++ b/experimental/sgl-router/tests/scripts/generate_deepseek_parity.py @@ -68,7 +68,7 @@ def main(): fixture_path = ROOT / (family + ".json") fixture = json.loads(fixture_path.read_text()) for case in fixture["cases"]: - server._dsv4_reasoning_effort_profile = case["profile"] + server._dsv4_reasoning_effort_profile = case.get("profile") tok.texts.clear() request = ChatCompletionRequest(**copy.deepcopy(case["request"])) # _convert_to_internal_request: kwargs effort replaces the request effort. From 00491cf532c268d8325fd667f066ee71ef1ad040 Mon Sep 17 00:00:00 2001 From: Kan Wu Date: Fri, 25 Sep 2026 06:47:25 +0000 Subject: [PATCH 2/5] [sgl-router] Scope forwarding by renderer: all text for DeepSeek-V4, guarded elsewhere, never for V4.1 Keep the per-shape guard for formatters that are not verified against SGLang, drop it only for the native DeepSeek-V4 renderer, and disable forwarding for DeepSeek-V4.1, whose effort mapping is stale against SGLang main. Co-Authored-By: Claude Opus 5.5 --- .../src/server/routes/chat/preparation.rs | 319 ++++++++++++++++-- .../src/tokenizer/chat_formatter.rs | 10 +- experimental/sgl-router/src/tokenizer/mod.rs | 45 ++- .../tests/proxy/roundrobin_input_ids.rs | 75 ++++ 4 files changed, 410 insertions(+), 39 deletions(-) diff --git a/experimental/sgl-router/src/server/routes/chat/preparation.rs b/experimental/sgl-router/src/server/routes/chat/preparation.rs index 7760b21a8c1d..2ed4d53f3697 100644 --- a/experimental/sgl-router/src/server/routes/chat/preparation.rs +++ b/experimental/sgl-router/src/server/routes/chat/preparation.rs @@ -9,6 +9,7 @@ use crate::policies::{has_caller_input_ids, request_tokens_for, RequestTokens}; use crate::server::app_context::AppContext; use crate::server::error::ApiError; use crate::server::metrics::{InputIdsForwarding, MetricsRegistry}; +use crate::tokenizer::ForwardingScope; use bytes::Bytes; use serde::de::IgnoredAny; use serde::Deserialize; @@ -28,7 +29,7 @@ pub(super) struct PreparedChatRequest { pub(super) input_token_count: usize, caller_set_rid: bool, fans_out: bool, - can_forward_input_ids: bool, + forwarding_scope: ForwardingScope, parsed_body: Option, sampling_defaults: Vec<(SamplingField, Number)>, } @@ -44,8 +45,12 @@ impl PreparedChatRequest { // Validate configured sampling rules and collect missing defaults for forwarding. let sampling_defaults = resolve_sampling_defaults(&ctx.config.model.sampling_overrides, &fields, &ctx.metrics)?; - let can_forward_input_ids = !ctx.config.model.disable_input_ids_forwarding - && ctx.tokenizers.has_chat_formatter(&model.0); + let forwarding_scope = if ctx.config.model.disable_input_ids_forwarding { + ForwardingScope::Never + } else { + ctx.tokenizers.forwarding_scope(&model.0) + }; + let can_forward_input_ids = forwarding_scope != ForwardingScope::Never; let needs_tokens = should_tokenize_request( can_forward_input_ids, policy_needs_request_tokens, @@ -73,7 +78,7 @@ impl PreparedChatRequest { input_token_count, caller_set_rid: fields.caller_set_rid, fans_out: requests_multiple_samples(&fields, &sampling_defaults), - can_forward_input_ids, + forwarding_scope, parsed_body, sampling_defaults, }) @@ -95,7 +100,7 @@ impl PreparedChatRequest { ) -> Result { // Routing tokens can replace engine tokenization only for supported chat templates. let forwarding = input_ids_forwarding( - self.can_forward_input_ids, + self.forwarding_scope, self.parsed_body.as_ref(), self.tokens.as_ref(), ); @@ -516,23 +521,76 @@ fn build_outgoing_body( Ok(Bytes::from(bytes)) } +/// Forward generated IDs only for request shapes verified against the engine. +/// The engine uses `input_ids` verbatim, bypassing its chat-template processing. +/// +/// Preserve caller-provided IDs. Exclude requests that may render differently +/// with dynamo-render: +/// - Non-leading system turns or consecutive users, which strict templates rewrite. +/// - Historical `reasoning_content`, which may be injected into message content. +/// - Tools and tool-call history, which the engine merges and normalizes +/// before rendering. +/// - Non-string or missing content, which the engine flattens or blanks. +/// - Template overrides, kwargs, reasoning controls, or task selection. +/// - Assistant continuations, whose final turn the engine handles separately. +/// +/// Matching model files and engine defaults are still required. Worker template +/// overrides and default kwargs cannot be inferred from the request. +/// `--disable-input-ids-forwarding` gates forwarding separately for such fleets. +fn can_forward_chat_tokens(value: &Value) -> bool { + if has_caller_input_ids(value) + || request_has_tools(value) + || request_has_non_text_content(value) + || request_has_reasoning_content(value) + || request_has_role_rewrites(value) + { + return false; + } + // Request controls whose rendering has not been verified against the engine. + for key in [ + "chat_template", + "chat_template_kwargs", + "reasoning", + "reasoning_effort", + "task", + ] { + if value.get(key).is_some_and(|v| !v.is_null()) { + return false; + } + } + if value + .get("continue_final_message") + .and_then(|v| v.as_bool()) + == Some(true) + { + return false; + } + !last_message_is_assistant(value) +} + /// Whether router-rendered `input_ids` replace engine tokenization. /// -/// Every text chat is eligible; multimodal chats and caller-provided IDs are not. -/// An eligible chat without chat-rendered tokens is a failed offload. +/// Only chats with forwarding enabled that pass the forwarding guard are +/// eligible; an eligible chat without chat-rendered tokens is a failed offload. +/// Multimodal chats are reported apart from other guard exclusions; +/// `AllText` models skip the guard. fn input_ids_forwarding( - can_forward_input_ids: bool, + scope: ForwardingScope, request_value: Option<&Value>, request_tokens: Option<&RequestTokens>, ) -> InputIdsForwarding { - if !can_forward_input_ids { + if scope == ForwardingScope::Never { return InputIdsForwarding::Disabled; } if request_value.is_some_and(request_has_multimodal_content) { return InputIdsForwarding::IneligibleMultimodal; } let eligible = request_value.is_some_and(|v| { - v.get("messages").is_some_and(|m| m.is_array()) && !has_caller_input_ids(v) + v.get("messages").is_some_and(|m| m.is_array()) + && match scope { + ForwardingScope::AllText => !has_caller_input_ids(v), + _ => can_forward_chat_tokens(v), + } }); if !eligible { InputIdsForwarding::Ineligible @@ -543,8 +601,86 @@ fn input_ids_forwarding( } } +/// Whether the final chat message has `role: "assistant"` (a prefix / +/// continuation turn the engine's template path special-cases). +fn last_message_is_assistant(value: &Value) -> bool { + value + .get("messages") + .and_then(|m| m.as_array()) + .and_then(|msgs| msgs.last()) + .and_then(|m| m.get("role")) + .and_then(|r| r.as_str()) + == Some("assistant") +} + +/// Tool schemas and tool-call history require engine normalization before +/// rendering: the engine merges message-level `tools` into the template's tools +/// and parses `tool_calls` arguments; dynamo-render does neither the same way. +fn request_has_tools(value: &Value) -> bool { + let nonempty = |v: &Value| match v { + Value::Array(a) => !a.is_empty(), + Value::Null => false, + _ => true, + }; + if ["tools", "functions"] + .iter() + .any(|key| value.get(key).is_some_and(nonempty)) + { + return true; + } + value + .get("messages") + .and_then(|m| m.as_array()) + .is_some_and(|messages| { + messages.iter().any(|message| { + message["role"] == "tool" + || ["tools", "tool_calls", "function_call"] + .iter() + .any(|key| message.get(key).is_some_and(nonempty)) + }) + }) +} + +/// dynamo-render may inject historical reasoning into content the engine leaves unchanged. +fn request_has_reasoning_content(value: &Value) -> bool { + value + .get("messages") + .and_then(|messages| messages.as_array()) + .is_some_and(|messages| { + messages.iter().any(|message| { + message + .get("reasoning_content") + .is_some_and(|v| !v.is_null()) + }) + }) +} + +/// Message orders dynamo-render may rewrite for strict templates. +fn request_has_role_rewrites(value: &Value) -> bool { + let Some(messages) = value.get("messages").and_then(|v| v.as_array()) else { + return false; + }; + messages.iter().skip(1).any(|m| m["role"] == "system") + || messages + .windows(2) + .any(|pair| pair[0]["role"] == "user" && pair[1]["role"] == "user") +} + +/// Detect non-string or missing content, which requires engine tokenization: +/// the engine normalizes arrays and nulls differently from dynamo-render. +fn request_has_non_text_content(value: &Value) -> bool { + value + .get("messages") + .and_then(|m| m.as_array()) + .is_some_and(|msgs| { + msgs.iter() + .any(|m| !matches!(m.get("content"), Some(Value::String(_)))) + }) +} + /// Image, video, and audio content parts need the engine's multimodal -/// processor, so such chats can never carry `input_ids`. +/// processor, so such chats can never carry `input_ids`. Other non-string +/// parts (`refusal`, `thinking`, untyped) are ordinary guard exclusions. fn request_has_multimodal_content(value: &Value) -> bool { value .get("messages") @@ -714,6 +850,42 @@ mod tests { } } + #[test] + fn request_has_tools_detects_tools_and_functions() { + assert!(request_has_tools(&json!({"tools":[{"type":"function"}]}))); + assert!(request_has_tools(&json!({"functions":[{"name":"f"}]}))); + assert!(!request_has_tools(&json!({"tools":[]}))); + assert!(!request_has_tools(&json!({"messages":[]}))); + for message in [ + json!({"role":"system","content":"s","tools":[{"type":"function"}]}), + json!({"role":"assistant","content":"","tool_calls":[{"function":{"name":"f","arguments":"{}"}}]}), + ] { + assert!(request_has_tools(&json!({"messages":[message]}))); + } + } + + #[test] + fn request_has_non_text_content_detects_non_string_content() { + for content in [ + json!([{"type":"image_url","image_url":"x"}]), + json!([{"type":"text","text":"a"},{"type":"text","text":"b"}]), + Value::Null, + ] { + assert!( + request_has_non_text_content(&json!({ + "messages":[{"role":"user","content":"hi"},{"role":"assistant","content":content}] + })), + "content {content} must block" + ); + } + assert!(request_has_non_text_content(&json!({ + "messages":[{"role":"assistant","tool_calls":[]}] + }))); + assert!(!request_has_non_text_content(&json!({ + "messages":[{"role":"user","content":"hello"}] + }))); + } + #[test] fn request_has_multimodal_content_detects_media_parts() { for part in [ @@ -747,6 +919,95 @@ mod tests { } } + #[test] + fn reasoning_history_is_an_expected_forwarding_omission() { + let mut value = json!({"messages": [ + {"role":"user", "content":"hi"}, + {"role":"assistant", "content":"answer", "reasoning_content":"prior reasoning"}, + {"role":"user", "content":"next"} + ]}); + assert!(!can_forward_chat_tokens(&value)); + assert_eq!( + input_ids_forwarding(ForwardingScope::Guarded, Some(&value), None), + InputIdsForwarding::Ineligible + ); + value["messages"][1]["reasoning_content"] = Value::Null; + assert!(can_forward_chat_tokens(&value)); + value["messages"][1] + .as_object_mut() + .unwrap() + .remove("reasoning_content"); + assert!(can_forward_chat_tokens(&value)); + } + + #[test] + fn role_rewrites_are_expected_forwarding_omissions() { + for roles in [ + vec!["user", "user"], + vec!["system", "system", "user"], + vec!["user", "assistant", "system", "user"], + ] { + let messages: Vec<_> = roles + .iter() + .map(|role| json!({"role": role, "content": "text"})) + .collect(); + let value = json!({"messages": messages}); + assert!(!can_forward_chat_tokens(&value), "{roles:?}"); + assert_eq!( + input_ids_forwarding(ForwardingScope::Guarded, Some(&value), None), + InputIdsForwarding::Ineligible + ); + } + assert!(can_forward_chat_tokens(&json!({"messages": [ + {"role": "system", "content": "instructions"}, + {"role": "user", "content": "hi"}, + {"role": "assistant", "content": "hello"}, + {"role": "user", "content": "next"} + ]}))); + } + + #[test] + fn can_forward_chat_tokens_allows_plain_text_chat() { + assert!(can_forward_chat_tokens(&json!({ + "messages": [{"role": "user", "content": "hello"}] + }))); + } + + #[test] + fn can_forward_chat_tokens_blocks_unreplicated_signals() { + let blockers = [ + json!({"messages":[{"role":"user","content":"hi"}],"input_ids":[7, 8]}), + json!({"messages":[{"role":"user","content":"hi"}],"input_ids":"bad"}), + json!({"messages":[{"role":"user","content":"hi"}],"tools":[{"type":"function"}]}), + json!({"messages":[{"role":"user","content":[{"type":"image_url","image_url":"x"}]}]}), + json!({"messages":[{"role":"user","content":"hi"}],"chat_template":"{{ custom }}"}), + json!({"messages":[{"role":"user","content":"hi"}],"chat_template_kwargs":{"enable_thinking":true}}), + json!({"messages":[{"role":"user","content":"hi"}],"reasoning_effort":"high"}), + json!({"messages":[{"role":"user","content":"hi"}],"reasoning":{"enabled":true}}), + json!({"messages":[{"role":"user","content":"hi"}],"task":"generate"}), + json!({"messages":[{"role":"user","content":"hi"}],"continue_final_message":true}), + json!({"messages":[{"role":"user","content":"hi"},{"role":"assistant","content":"partial"}]}), + ]; + for b in blockers { + assert!( + !can_forward_chat_tokens(&b), + "must NOT forward input_ids for: {b}" + ); + } + } + + #[test] + fn can_forward_chat_tokens_ignores_null_and_false_fields() { + assert!(can_forward_chat_tokens(&json!({ + "messages": [{"role": "user", "content": "hi"}], + "input_ids": null, + "chat_template": null, + "reasoning_effort": null, + "chat_template_kwargs": null, + "continue_final_message": false + }))); + } + #[test] fn input_ids_forwarding_outcome_per_request() { use InputIdsForwarding::*; @@ -760,27 +1021,35 @@ mod tests { let text_parts = json!({"messages":[{"role":"user","content":[{"type":"text","text":"hi"}]}]}); let caller_ids = json!({"messages":[{"role":"user","content":"hi"}], "input_ids":[7]}); - for (enabled, value, rendered, expected) in [ - (true, Some(&chat), Some(true), Forwarded), - (true, Some(&chat), Some(false), TokenizeFailed), - (true, Some(&chat), None, TokenizeFailed), - (false, Some(&chat), Some(true), Disabled), - (true, Some(&tools), Some(true), Forwarded), - (true, Some(&caller_ids), Some(false), Ineligible), - (true, Some(&prompt), None, Ineligible), - (true, None, None, Ineligible), - (true, Some(&image), None, IneligibleMultimodal), - (false, Some(&image), None, Disabled), - (true, Some(&text_parts), Some(true), Forwarded), + let (never, guarded, all) = ( + ForwardingScope::Never, + ForwardingScope::Guarded, + ForwardingScope::AllText, + ); + for (scope, value, rendered, expected) in [ + (guarded, Some(&chat), Some(true), Forwarded), + (guarded, Some(&chat), Some(false), TokenizeFailed), + (guarded, Some(&chat), None, TokenizeFailed), + (never, Some(&chat), Some(true), Disabled), + (guarded, Some(&tools), Some(true), Ineligible), + (guarded, Some(&prompt), None, Ineligible), + (guarded, None, None, Ineligible), + (guarded, Some(&image), None, IneligibleMultimodal), + (never, Some(&image), None, Disabled), + (guarded, Some(&text_parts), None, Ineligible), + (all, Some(&tools), Some(true), Forwarded), + (all, Some(&text_parts), Some(true), Forwarded), + (all, Some(&image), None, IneligibleMultimodal), + (all, Some(&caller_ids), Some(false), Ineligible), ] { let tokens = rendered.map(|rendered_from_chat| RequestTokens { ids: vec![1, 2, 3], rendered_from_chat, }); assert_eq!( - input_ids_forwarding(enabled, value, tokens.as_ref()), + input_ids_forwarding(scope, value, tokens.as_ref()), expected, - "enabled={enabled}, request={value:?}, rendered={rendered:?}" + "scope={scope:?}, request={value:?}, rendered={rendered:?}" ); } } diff --git a/experimental/sgl-router/src/tokenizer/chat_formatter.rs b/experimental/sgl-router/src/tokenizer/chat_formatter.rs index 2c786c8f4c6a..81220522bef4 100644 --- a/experimental/sgl-router/src/tokenizer/chat_formatter.rs +++ b/experimental/sgl-router/src/tokenizer/chat_formatter.rs @@ -231,9 +231,13 @@ impl ChatFormatter { }) } - /// Whether router rendering is fixture-verified against SGLang for every text chat. - pub fn forwarding_verified(&self) -> bool { - matches!(self.deepseek, Some(super::deepseek::Encoder::V4(_))) + /// DeepSeek-V4 is fixture-verified against SGLang for every text chat; V4.1 is stale. + pub fn forwarding_scope(&self) -> super::ForwardingScope { + match self.deepseek { + Some(super::deepseek::Encoder::V4(_)) => super::ForwardingScope::AllText, + Some(super::deepseek::Encoder::V41) => super::ForwardingScope::Never, + None => super::ForwardingScope::Guarded, + } } /// Apply the workers' `--default-chat-template-kwargs`; they fill keys the diff --git a/experimental/sgl-router/src/tokenizer/mod.rs b/experimental/sgl-router/src/tokenizer/mod.rs index f607b61dcbfd..fb7df3c5c275 100644 --- a/experimental/sgl-router/src/tokenizer/mod.rs +++ b/experimental/sgl-router/src/tokenizer/mod.rs @@ -13,6 +13,17 @@ use dynamo_tokenizers::Tokenizer; use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::Arc; +/// Which chats may carry router-rendered `input_ids`, by how well the model's +/// renderer is verified against SGLang. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum ForwardingScope { + Never, + /// Only request shapes the per-shape guard allows. + Guarded, + /// Every non-multimodal chat. + AllText, +} + /// A model's chat formatter plus its fallback-logging state. struct ChatFormatterEntry { formatter: ChatFormatter, @@ -82,19 +93,25 @@ impl TokenizerRegistry { tracing::info!(model = %m.id, "router-generated input_ids forwarding disabled; workers tokenize messages; \ routing tokenization remains available"); - } else if let Some(entry) = me.formatters.get(&m.id) { - if entry.formatter.forwarding_verified() { - tracing::info!(model = %m.id, + } else { + match me.forwarding_scope(&m.id) { + ForwardingScope::AllText => tracing::info!(model = %m.id, "router-generated input_ids forwarding enabled for all text chats; requires the \ workers' model files, --default-chat-template-kwargs, SGLANG_DEFAULT_THINKING, \ - and SGLANG_DSV4_REASONING_EFFORT"); - } else { - tracing::warn!(model = %m.id, - "UNVERIFIED input_ids forwarding: router rendering is verified against SGLang \ - only for DeepSeek-V4; this model's chats may be forwarded with prompts that \ - differ from what the workers would render (tools, reasoning history, content \ - parts, strict templates). Pass --disable-input-ids-forwarding unless you have \ - verified parity for this model"); + and SGLANG_DSV4_REASONING_EFFORT"), + ForwardingScope::Guarded => tracing::warn!(model = %m.id, + "UNVERIFIED input_ids forwarding: router rendering is verified against SGLang only \ + for DeepSeek-V4, so this model forwards only guarded request shapes (plain text \ + chat). Requires the workers' model files and --default-chat-template-kwargs; \ + worker parser overrides, content-format detection, and conversation-template stop \ + strings are not replicated. Pass --disable-input-ids-forwarding unless you have \ + verified parity for this model"), + ForwardingScope::Never if me.has_chat_formatter(&m.id) => { + tracing::warn!(model = %m.id, + "input_ids forwarding disabled: the DeepSeek-V4.1 renderer is not verified against \ + current SGLang; workers tokenize messages") + } + ForwardingScope::Never => {} } } Ok(me) @@ -110,6 +127,12 @@ impl TokenizerRegistry { self.formatters.contains_key(model_id) } + pub fn forwarding_scope(&self, model_id: &str) -> ForwardingScope { + self.formatters + .get(model_id) + .map_or(ForwardingScope::Never, |e| e.formatter.forwarding_scope()) + } + /// Render with dynamo-render and tokenize; return `None` when unavailable or unsuccessful. pub fn encode_chat(&self, model_id: &str, request: &serde_json::Value) -> Option> { // Clone the Arc and drop the DashMap guard before the CPU-bound diff --git a/experimental/sgl-router/tests/proxy/roundrobin_input_ids.rs b/experimental/sgl-router/tests/proxy/roundrobin_input_ids.rs index 634d885bfd94..3e96426a353c 100644 --- a/experimental/sgl-router/tests/proxy/roundrobin_input_ids.rs +++ b/experimental/sgl-router/tests/proxy/roundrobin_input_ids.rs @@ -351,6 +351,81 @@ async fn successful_forward_does_not_emit_ingress_tokenize_error() { ); } +#[tokio::test] +async fn reasoning_history_preserves_messages_without_forwarding_ids() { + let (_dir, cfg) = template_config(json!({ + "chat_template": "{% for m in messages %}{{ m.role }}:{{ m.content }};{% endfor %}" + })); + let mock = MockWorker::start(vec![]).await; + let ctx = build_ctx_with_config(mock.url.clone(), cfg); + let mut request = json!({"model": MODEL, "messages": [ + {"role":"user", "content":"hi"}, + {"role":"assistant", "content":"answer", "reasoning_content":"prior reasoning"}, + {"role":"user", "content":"next"} + ]}); + assert!(!ctx + .tokenizers + .encode_chat(MODEL, &request) + .unwrap() + .is_empty()); + assert_forwarded_unchanged(&ctx, &mock, &request).await; + + request["messages"][1] + .as_object_mut() + .unwrap() + .remove("reasoning_content"); + assert_eq!(send(ctx, request).await, StatusCode::OK); + assert!(captured(&mock).get("input_ids").is_some()); +} + +/// Strict-template rewrites are used for routing only; the engine gets the original turns. +#[tokio::test] +async fn role_rewrites_preserve_messages_without_forwarding_ids() { + let template = concat!( + "{%- set ns = namespace(prev='') -%}", + "{%- for m in messages -%}", + "{%- if m.role == 'system' and not loop.first -%}", + "{{ raise_exception('System message must be first.') }}", + "{%- endif -%}", + "{%- if m.role == 'user' and ns.prev == 'user' -%}", + "{{ raise_exception('Conversation roles must alternate.') }}", + "{%- endif -%}", + "{{ m.role }}:{{ m.content }};", + "{%- set ns.prev = m.role -%}", + "{%- endfor -%}" + ); + let (_dir, cfg) = template_config(json!({ + "chat_template": template, "sp_model_kwargs": {"enable_sampling": false} + })); + let mock = MockWorker::start(vec![]).await; + let ctx = build_ctx_with_config(mock.url.clone(), cfg); + for roles in [ + vec!["user", "user"], + vec!["system", "system", "user"], + vec!["user", "assistant", "system", "user"], + ] { + let messages: Vec<_> = roles + .iter() + .map(|role| json!({"role": role, "content": "text"})) + .collect(); + let request = json!({"model": MODEL, "messages": messages}); + assert!(!ctx + .tokenizers + .encode_chat(MODEL, &request) + .unwrap() + .is_empty()); + assert_forwarded_unchanged(&ctx, &mock, &request).await; + } + let request = json!({"model": MODEL, "messages": [ + {"role": "system", "content": "instructions"}, + {"role": "user", "content": "hi"}, + {"role": "assistant", "content": "hello"}, + {"role": "user", "content": "next"} + ]}); + assert_eq!(send(ctx, request).await, StatusCode::OK); + assert!(captured(&mock).get("input_ids").is_some()); +} + #[path = "../fixtures/kimi_k3.rs"] mod kimi_fixture; From ef1f3c7ad8b5a1cef95ccfdbcadba48fa9ec279b Mon Sep 17 00:00:00 2001 From: Kan Wu Date: Fri, 25 Sep 2026 18:05:17 +0000 Subject: [PATCH 3/5] [sgl-router] Document that unverified renderers book as disabled DeepSeek-V4.1 now has ForwardingScope::Never, so its chats book as outcome="disabled" even without --disable-input-ids-forwarding. Say so in the metric doc, and note that AllText models book ineligible only for caller input_ids. Co-Authored-By: Claude Opus 5.5 --- experimental/sgl-router/src/server/metrics.rs | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/experimental/sgl-router/src/server/metrics.rs b/experimental/sgl-router/src/server/metrics.rs index abdb3f1cf37e..02f68b540eb8 100644 --- a/experimental/sgl-router/src/server/metrics.rs +++ b/experimental/sgl-router/src/server/metrics.rs @@ -90,11 +90,13 @@ //! //! - `forwarded` — router-rendered `input_ids` replaced engine tokenization. //! - `disabled` — forwarding is off for the model (`--disable-input-ids-forwarding`, -//! or no chat formatter). +//! no chat formatter, or a renderer not verified against SGLang, such as +//! DeepSeek-V4.1). //! - `ineligible_multimodal` — the chat carries image, video, or audio content //! parts, which only the engine's multimodal processor can tokenize. //! - `ineligible` — the forwarding guard excluded some other request shape -//! (tools, non-string content, caller `input_ids`, template controls, ...). +//! (tools, non-string content, caller `input_ids`, template controls, ...; +//! only caller `input_ids` on models that forward all text chats). //! - `tokenize_failed` — eligible, but ingress rendering failed (the same //! requests `sgl_router_ingress_tokenize_errors_total` counts). //! From 8dadf1cf6beafcf27a67dd7df924ee30933d0209 Mon Sep 17 00:00:00 2001 From: Kan Wu Date: Sat, 26 Sep 2026 21:06:49 +0000 Subject: [PATCH 4/5] docs(router): update V4.1 forwarding comment after restack --- experimental/sgl-router/src/tokenizer/chat_formatter.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/experimental/sgl-router/src/tokenizer/chat_formatter.rs b/experimental/sgl-router/src/tokenizer/chat_formatter.rs index 81220522bef4..9647cdef64d2 100644 --- a/experimental/sgl-router/src/tokenizer/chat_formatter.rs +++ b/experimental/sgl-router/src/tokenizer/chat_formatter.rs @@ -231,7 +231,7 @@ impl ChatFormatter { }) } - /// DeepSeek-V4 is fixture-verified against SGLang for every text chat; V4.1 is stale. + /// DeepSeek-V4 is fixture-verified against SGLang for every text chat; V4.1 stays disabled. pub fn forwarding_scope(&self) -> super::ForwardingScope { match self.deepseek { Some(super::deepseek::Encoder::V4(_)) => super::ForwardingScope::AllText, From e4c02f385ee86fbc6deec145f65a5204e9ee5308 Mon Sep 17 00:00:00 2001 From: Shangming Cai Date: Mon, 28 Sep 2026 18:35:54 +0800 Subject: [PATCH 5/5] [sgl-router] README: document renderer-scoped input_ids forwarding (#41558) --- experimental/sgl-router/README.md | 30 +++++++++++++++++++----------- 1 file changed, 19 insertions(+), 11 deletions(-) diff --git a/experimental/sgl-router/README.md b/experimental/sgl-router/README.md index 528103288703..030334126803 100644 --- a/experimental/sgl-router/README.md +++ b/experimental/sgl-router/README.md @@ -198,14 +198,20 @@ prefix queries match the blocks the engine caches. Models the engine encodes in code but dynamo-render cannot tokenize here (Inkling) route via raw prompt text, as does any model whose template fails to load or render. -Plain text chat requests (string `content`, no tools, no template kwargs or -reasoning controls or historical `reasoning_content`, no assistant continuation, -no consecutive users or non-leading system turns) additionally forward the -rendered tokens to the engine as `input_ids`, retaining the original messages, -so the engine skips re-tokenizing. Every other request shape is rendered for -routing only: the router renders with dynamo-render and does not replicate -SGLang's request normalization, so forwarding is enabled shape by shape as -parity is verified. Use matching model files on the router and workers, and set +Some chats additionally forward the rendered tokens to the engine as +`input_ids`, retaining the original messages, so the engine skips +re-tokenizing. How many depends on the model's renderer. DeepSeek-V4's native +encoder is fixture-verified against SGLang's request normalization, so it +forwards every chat except multimodal ones and those with caller-provided +`input_ids`. Renderers without that verification (HF Jinja templates, Kimi-K3) +forward only plain text chat requests (string `content`, no tools, no template +kwargs or reasoning controls or historical `reasoning_content`, no assistant +continuation, no consecutive users or non-leading system turns) and warn +`UNVERIFIED` at startup; every other request shape is rendered for routing +only. DeepSeek-V4.1 forwards nothing — its renderer is not verified against +current SGLang — while routing tokenization keeps working. + +Use matching model files on the router and workers, and set the same `--default-chat-template-kwargs`, `SGLANG_DEFAULT_THINKING`, `SGLANG_DSV4_REASONING_EFFORT`, and `SGLANG_DSV41_REASONING_EFFORT` on both. The router reads these render defaults from its own configuration and environment; @@ -246,9 +252,11 @@ official/preview effort profile is detected from the checkpoint's and are regenerated by `tests/scripts/generate_deepseek_parity.py`. V4.1 Flash uses Dynamo's separate V4.1 encoder with SGLang's numeric reasoning -budgets, tool payloads, and `<|System|>` markers. Developer messages and media -are left to the worker (the pinned encoder renders them differently), and a -non-default `SGLANG_DSV41_REASONING_EFFORT` needs the forwarding precautions above. +budgets, tool payloads, and `<|System|>` markers — for routing tokenization +only, since V4.1 never forwards `input_ids`. Developer messages and media are +left to the worker (the pinned encoder renders them differently), and a +non-default `SGLANG_DSV41_REASONING_EFFORT` still matters for cache-aware +routing-hash parity. ## Kimi-K3