From 893bbabfc969f99853f32e49a9868fa94d60525f Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Thu, 4 Jun 2026 05:38:59 -0700 Subject: [PATCH 01/28] Optimize TokenSpeed multimodal tensor transport Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- .../smg_grpc_servicer/tokenspeed/servicer.py | 166 ++++++++++- model_gateway/src/routers/grpc/multimodal.rs | 280 ++++++++++++++++-- .../src/routers/grpc/proto_wrapper.rs | 260 ++++++++++++++-- 3 files changed, 663 insertions(+), 43 deletions(-) diff --git a/grpc_servicer/smg_grpc_servicer/tokenspeed/servicer.py b/grpc_servicer/smg_grpc_servicer/tokenspeed/servicer.py index d0f9090ac8..91fcd1f2ac 100644 --- a/grpc_servicer/smg_grpc_servicer/tokenspeed/servicer.py +++ b/grpc_servicer/smg_grpc_servicer/tokenspeed/servicer.py @@ -35,6 +35,7 @@ MultimodalDataItem, MultimodalInputs, ) +from tokenspeed.runtime.multimodal.shm_transport import ShmTensorHandle from tokenspeed.runtime.pd.kv_events import KVEventBatch from smg_grpc_servicer.kv_events import endpoint_for_rank, stream_kv_events @@ -57,6 +58,14 @@ "true", "yes", ) +LOG_MM_TIMING = os.getenv("TOKENSPEED_LOG_MM_TIMING", "").lower() in ( + "1", + "true", + "yes", +) +UNLINK_MM_SHM_AFTER_READ = os.getenv( + "TOKENSPEED_UNLINK_MM_SHM_AFTER_READ", "1" +).lower() not in ("0", "false", "no") def _lazy_generate_req_input(): @@ -141,7 +150,18 @@ async def Generate( logger.info("Generate request %s (stream=%s)", rid, request.stream) try: + build_started = time.perf_counter() req_obj = self._build_generate_req(request) + if LOG_MM_TIMING: + has_mm = ( + getattr(req_obj, "precomputed_multimodal_inputs", None) is not None + ) + logger.info( + "mm_timing generate_build_ms rid=%s elapsed=%.3f has_mm=%s", + rid, + (time.perf_counter() - build_started) * 1000, + has_mm, + ) except ValueError as e: await context.abort(grpc.StatusCode.INVALID_ARGUMENT, str(e)) return @@ -160,7 +180,16 @@ async def Generate( aborted = False try: + engine_started = time.perf_counter() + first_output = True async for output in self.async_llm.generate_request(req_obj): + if LOG_MM_TIMING and first_output: + first_output = False + logger.info( + "mm_timing generate_first_output_ms rid=%s elapsed=%.3f", + rid, + (time.perf_counter() - engine_started) * 1000, + ) # Non-streaming n>1 emits a list of final dicts in one yield. # Pre-scan for aborts so we don't yield partial successes # before raising on a later aborted choice. @@ -923,32 +952,50 @@ def _mm_inputs_from_itemized_proto( items = [] im_token_id = None video_token_id = None + total_started = time.perf_counter() if LOG_MM_TIMING else None for item_proto in mm_inputs.items: + item_started = time.perf_counter() if LOG_MM_TIMING else None modality = self._modality_from_proto(item_proto.modality) if not item_proto.HasField("encoder_input"): raise ValueError("MultimodalItem must include encoder_input") - feature = self._tensor_from_proto(item_proto.encoder_input, cast_to=model_dtype) + feature_started = time.perf_counter() if LOG_MM_TIMING else None + feature = self._feature_from_proto( + item_proto.encoder_input, cast_to=model_dtype + ) + feature_elapsed_ms = ( + (time.perf_counter() - feature_started) * 1000 + if feature_started is not None + else None + ) if LOG_MM_TENSOR_DATA: encoder_input = item_proto.encoder_input payload = encoder_input.WhichOneof("payload") inline_nbytes = len(encoder_input.inline) if payload == "inline" else None logger.info( "Multimodal encoder_input received: modality=%s proto_dtype=%s " - "shape=%s payload=%s inline_nbytes=%s torch_dtype=%s cast_to=%s", + "shape=%s payload=%s inline_nbytes=%s feature_type=%s " + "torch_dtype=%s cast_to=%s", modality, encoder_input.dtype, list(encoder_input.shape), payload, inline_nbytes, + type(feature).__name__, feature.dtype, model_dtype, ) + model_started = time.perf_counter() if LOG_MM_TIMING else None model_specific_data = { name: self._tensor_from_proto(tensor_data, cast_to=model_dtype) for name, tensor_data in item_proto.model_specific_tensors.items() } + model_elapsed_ms = ( + (time.perf_counter() - model_started) * 1000 + if model_started is not None + else None + ) self._validate_item_tensor_consistency(modality, model_specific_data) if not item_proto.placeholders: @@ -968,6 +1015,23 @@ def _mm_inputs_from_itemized_proto( mm_item.set_pad_value() items.append(mm_item) + if LOG_MM_TIMING and item_started is not None: + encoder_input = item_proto.encoder_input + logger.info( + "mm_timing item_build_ms modality=%s elapsed=%.3f " + "feature_ms=%.3f model_specific_ms=%.3f payload=%s " + "proto_dtype=%s shape=%s feature_type=%s tensors=%s", + modality.name, + (time.perf_counter() - item_started) * 1000, + feature_elapsed_ms, + model_elapsed_ms, + encoder_input.WhichOneof("payload"), + encoder_input.dtype, + list(encoder_input.shape), + type(feature).__name__, + sorted(model_specific_data.keys()), + ) + if item_proto.HasField("placeholder_token_id"): placeholder_token_id = int(item_proto.placeholder_token_id) if modality == Modality.IMAGE: @@ -981,6 +1045,12 @@ def _mm_inputs_from_itemized_proto( if not items: raise ValueError("MultimodalInputs.items is empty") + if LOG_MM_TIMING and total_started is not None: + logger.info( + "mm_timing mm_inputs_build_ms items=%d elapsed=%.3f", + len(items), + (time.perf_counter() - total_started) * 1000, + ) return MultimodalInputs( mm_items=items, im_token_id=im_token_id, @@ -1060,17 +1130,107 @@ def _tensor_from_proto( return t.to(cast_to) return t.clone() + @staticmethod + def _feature_from_proto( + tensor_data: tokenspeed_scheduler_pb2.TensorData, + cast_to: torch.dtype | None = None, + ) -> torch.Tensor | ShmTensorHandle: + """Reconstruct a feature tensor, preserving SHM handles when possible. + + ``MultimodalInputs.publish_shm_features`` is a no-op for existing + ``ShmTensorHandle`` values, so the scheduler worker can consume the + upstream SMG segment directly instead of the gRPC frontend first + materializing bytes and cloning a CPU tensor. + """ + if tensor_data.WhichOneof("payload") != "shm": + return TokenSpeedSchedulerServicer._tensor_from_proto( + tensor_data, cast_to=cast_to + ) + + dtype = TokenSpeedSchedulerServicer._torch_dtype_from_proto(tensor_data.dtype) + if cast_to is not None and dtype != cast_to and torch.is_floating_point( + torch.empty((), dtype=dtype) + ): + return TokenSpeedSchedulerServicer._tensor_from_proto( + tensor_data, cast_to=cast_to + ) + + shm = tensor_data.shm + if shm.offset != 0: + return TokenSpeedSchedulerServicer._tensor_from_proto( + tensor_data, cast_to=cast_to + ) + + shape = tuple(int(dim) for dim in tensor_data.shape) + expected = int(np.prod(shape, dtype=np.int64)) * torch.empty( + (), dtype=dtype + ).element_size() + if int(shm.nbytes) != expected: + raise ValueError( + f"TensorData.shm byte length mismatch for dtype={tensor_data.dtype}, " + f"shape={list(shape)}: expected {expected}, got {int(shm.nbytes)}" + ) + + name = TokenSpeedSchedulerServicer._validated_shm_name(shm.name) + return ShmTensorHandle(shm_name=name, shape=shape, dtype=dtype) + @staticmethod def _tensor_payload_bytes(tensor_data: tokenspeed_scheduler_pb2.TensorData) -> bytes: payload = tensor_data.WhichOneof("payload") if payload == "inline": return bytes(tensor_data.inline) if payload == "shm": - raise ValueError("TensorData.shm payload is not implemented yet") + return TokenSpeedSchedulerServicer._tensor_payload_bytes_from_shm( + tensor_data.shm + ) if payload == "remote": raise ValueError("TensorData.remote payload is not implemented yet") raise ValueError("TensorData payload is required") + @staticmethod + def _tensor_payload_bytes_from_shm( + shm_handle: tokenspeed_scheduler_pb2.ShmHandle, + ) -> bytes: + name = TokenSpeedSchedulerServicer._validated_shm_name(shm_handle.name) + + path = os.path.join("/dev/shm", name) + fd = None + try: + fd = os.open(path, os.O_RDONLY) + raw = os.pread(fd, int(shm_handle.nbytes), int(shm_handle.offset)) + finally: + if fd is not None: + os.close(fd) + if fd is not None and UNLINK_MM_SHM_AFTER_READ: + try: + os.unlink(path) + except FileNotFoundError: + pass + + if len(raw) != int(shm_handle.nbytes): + raise ValueError( + f"TensorData.shm byte length mismatch for name={shm_handle.name!r}: " + f"expected {int(shm_handle.nbytes)}, got {len(raw)}" + ) + return raw + + @staticmethod + def _validated_shm_name(name: str) -> str: + name = name.lstrip("/") + if not name or "/" in name or name in (".", "..") or "\x00" in name: + raise ValueError(f"Invalid TensorData.shm name: {name!r}") + return name + + @staticmethod + def _torch_dtype_from_proto(dtype: str) -> torch.dtype: + if dtype == "bfloat16": + return torch.bfloat16 + if dtype == "float16": + return torch.float16 + if dtype == "float32": + return torch.float32 + raise ValueError(f"Unsupported TensorData dtype for SHM feature: {dtype!r}") + @staticmethod def _torch_dtype_to_proto(dtype: torch.dtype | None) -> str: if dtype is torch.bfloat16: diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index e322930b22..0a2eb88f48 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -11,8 +11,11 @@ use std::{ collections::HashMap, + io::Write, + mem::size_of, path::Path, sync::{Arc, OnceLock}, + time::Instant, }; use anyhow::{Context, Result}; @@ -30,14 +33,15 @@ use openai_protocol::{ common::ContentPart, messages::{ImageSource, InputContent, InputContentBlock, InputMessage, Role}, }; -use tracing::{debug, warn}; +use tracing::{debug, info, warn}; use crate::routers::grpc::{ client::GrpcClient, context::WorkerSelection, proto_wrapper::{ + tokenspeed_shm_transport_enabled_for_bytes, write_tokenspeed_shm_with, SglangMultimodalData, TensorBytes, TokenSpeedModality, TokenSpeedMultimodalData, - TokenSpeedMultimodalItem, TrtllmMultimodalData, VllmMultimodalData, + TokenSpeedMultimodalItem, TokenSpeedTensorBytes, TrtllmMultimodalData, VllmMultimodalData, }, MultimodalData, }; @@ -62,6 +66,12 @@ pub struct MultimodalConfigRegistry { configs: DashMap>, } +fn log_mm_timing_enabled() -> bool { + std::env::var("SMG_LOG_MM_TIMING") + .map(|value| matches!(value.to_ascii_lowercase().as_str(), "1" | "true" | "yes")) + .unwrap_or(false) +} + impl MultimodalConfigRegistry { pub(crate) fn new() -> Self { Self { @@ -546,6 +556,9 @@ async fn process_multimodal_parts( tokenizer_id: &str, tokenizer_source: &str, ) -> Result { + let log_timing = log_mm_timing_enabled(); + let total_started = Instant::now(); + let media_started = Instant::now(); let mut tracker = AsyncMultiModalTracker::new(components.media_connector.clone()); for part in content_parts { @@ -587,6 +600,7 @@ async fn process_multimodal_parts( }) .unwrap_or_default(); + let media_elapsed_ms = media_started.elapsed().as_secs_f64() * 1000.0; let modality = match (images.is_empty(), videos.is_empty()) { (false, true) => Modality::Image, (true, false) => Modality::Video, @@ -627,6 +641,7 @@ async fn process_multimodal_parts( } // Step 2: Resolve model spec and preprocess media. + let config_started = Instant::now(); let model_config = components .config_registry .get_or_load(tokenizer_id, tokenizer_source) @@ -644,6 +659,7 @@ async fn process_multimodal_parts( .model_registry .lookup(&metadata) .ok_or_else(|| anyhow::anyhow!("Multimodal not supported for model: {model_id}"))?; + let config_elapsed_ms = config_started.elapsed().as_secs_f64() * 1000.0; // Run CPU-intensive vision preprocessing on a blocking thread pool so it // doesn't block the tokio async runtime under concurrent load. @@ -661,6 +677,7 @@ async fn process_multimodal_parts( let images_for_preprocess = images.clone(); // cheap Arc refcount bumps let videos_for_preprocess = videos.clone(); // cheap Arc refcount bumps + let preprocess_started = Instant::now(); let preprocessed: PreprocessedEncoderInputs = tokio::task::spawn_blocking(move || { let processor = registry .find(&model_id_owned, model_type_owned.as_deref()) @@ -727,6 +744,7 @@ async fn process_multimodal_parts( }) .await .map_err(|e| anyhow::anyhow!("Preprocessing task panicked: {e}"))??; + let preprocess_elapsed_ms = preprocess_started.elapsed().as_secs_f64() * 1000.0; debug!( ?modality, @@ -736,6 +754,7 @@ async fn process_multimodal_parts( ); // Step 3: Compute prompt replacements and expand tokens. + let expansion_started = Instant::now(); let prompt_replacements = spec .prompt_replacements_for(&metadata, &preprocessed, modality) .map_err(|e| anyhow::anyhow!("Failed to compute prompt replacements: {e}"))?; @@ -775,6 +794,20 @@ async fn process_multimodal_parts( ?placeholder_token_id, "Token expansion complete" ); + let expansion_elapsed_ms = expansion_started.elapsed().as_secs_f64() * 1000.0; + let image_count = images.len(); + let video_count = videos.len(); + let video_frame_count = videos.first().map_or(0, |video| { + if !video.frames().is_empty() { + video.frames().len() + } else { + video + .rgb_video() + .map_or(0, |rgb_video| rgb_video.frames.len()) + } + }); + let original_tokens = token_ids.len(); + let expanded_tokens = expanded.token_ids.len(); // Step 4: Build lightweight intermediate (defers tensor serialization to assembly) let intermediate = MultimodalIntermediate::Precomputed(PrecomputedMultimodalIntermediate { @@ -789,6 +822,23 @@ async fn process_multimodal_parts( keep_on_cpu_keys: spec.keep_on_cpu_keys(), }); + if log_timing { + info!( + modality = ?modality, + image_count, + video_count, + video_frame_count, + media_fetch_decode_ms = media_elapsed_ms, + config_lookup_ms = config_elapsed_ms, + preprocess_ms = preprocess_elapsed_ms, + token_expand_ms = expansion_elapsed_ms, + total_ms = total_started.elapsed().as_secs_f64() * 1000.0, + original_tokens, + expanded_tokens, + "smg_mm_timing process_multimodal_parts" + ); + } + Ok(MultimodalOutput { expanded_token_ids: expanded.token_ids, intermediate, @@ -1008,6 +1058,8 @@ fn assemble_tokenspeed( intermediate: PrecomputedMultimodalIntermediate, workers: Option<&WorkerSelection>, ) -> Result { + let log_timing = log_mm_timing_enabled(); + let total_started = Instant::now(); // Use patch-only offsets when available and non-empty; fall back to full structural ranges. let encoder_input_dtype = tokenspeed_encoder_input_dtype(intermediate.modality, workers); let patch_offsets = intermediate @@ -1031,23 +1083,40 @@ fn assemble_tokenspeed( &intermediate.field_layouts, item_index, )?; - let (encoder_input, encoder_input_shape, encoder_input_dtype) = - serialize_array_as_dtype(&item_encoder_input, &encoder_input_dtype); + let encoder_input_started = Instant::now(); + let encoder_input = + serialize_array_as_tokenspeed_tensor(&item_encoder_input, &encoder_input_dtype); + let encoder_input_serialize_ms = encoder_input_started.elapsed().as_secs_f64() * 1000.0; + let model_specific_started = Instant::now(); let model_specific_tensors = serialize_model_specific_for_item( &intermediate.preprocessed.model_specific, &intermediate.field_layouts, item_index, )?; + let model_specific_serialize_ms = + model_specific_started.elapsed().as_secs_f64() * 1000.0; let mm_placeholders = placeholders_for_item(item_index, &intermediate.placeholders, &patch_offsets); let content_hash = content_hash_for_item(intermediate.modality, &intermediate, item_index); + if log_timing { + info!( + modality = ?modality, + item_index, + encoder_input_dtype = %encoder_input.dtype, + encoder_input_bytes = encoder_input.nbytes(), + encoder_input_shape = ?encoder_input.shape, + model_specific_tensor_count = model_specific_tensors.len(), + encoder_input_serialize_ms, + model_specific_serialize_ms, + "smg_mm_timing assemble_tokenspeed_item" + ); + } + Ok(TokenSpeedMultimodalItem { modality, encoder_input, - encoder_input_shape, - encoder_input_dtype, model_specific_tensors, placeholder_token_id: intermediate.placeholder_token_id, mm_placeholders, @@ -1056,6 +1125,15 @@ fn assemble_tokenspeed( }) .collect::>>()?; + if log_timing { + info!( + modality = ?modality, + item_count = items.len(), + total_ms = total_started.elapsed().as_secs_f64() * 1000.0, + "smg_mm_timing assemble_tokenspeed" + ); + } + Ok(TokenSpeedMultimodalData { items }) } @@ -1266,6 +1344,145 @@ fn serialize_array(encoder_input: &ArrayD) -> (Vec, Vec) { (encoder_bytes, array_shape(encoder_input)) } +/// Serialize encoder input to the requested wire dtype. +fn serialize_array_as_tokenspeed_tensor( + encoder_input: &ArrayD, + dtype: &str, +) -> TokenSpeedTensorBytes { + let dtype = match canonical_float_dtype(dtype).as_deref() { + Some("float32") => "float32".to_string(), + Some("bfloat16") => "bfloat16".to_string(), + Some("float16") => "float16".to_string(), + _ => { + warn!( + dtype, + "Unsupported TokenSpeed encoder input dtype; falling back to float32" + ); + "float32".to_string() + } + }; + let shape = array_shape(encoder_input); + let nbytes = encoder_input.len() + * match dtype.as_str() { + "float32" => size_of::(), + "bfloat16" | "float16" => size_of::(), + _ => unreachable!("canonical dtype is constrained above"), + }; + + if tokenspeed_shm_transport_enabled_for_bytes(nbytes) { + let started = Instant::now(); + match write_tokenspeed_shm_with(nbytes, |file| { + write_array_as_dtype(file, encoder_input, &dtype) + }) { + Ok(handle) => { + if log_mm_timing_enabled() { + info!( + nbytes, + elapsed_ms = started.elapsed().as_secs_f64() * 1000.0, + "smg_mm_timing tokenspeed_shm_write_direct" + ); + } + return TokenSpeedTensorBytes::shm(handle, shape, dtype); + } + Err(error) => { + warn!( + ?error, + nbytes, + dtype = %dtype, + "Failed to write TokenSpeed encoder input directly to SHM; falling back to bytes path" + ); + } + } + } + + let (data, shape, dtype) = serialize_array_as_dtype(encoder_input, &dtype); + TokenSpeedTensorBytes::bytes(data, shape, dtype) +} + +fn write_array_as_dtype( + writer: &mut impl Write, + encoder_input: &ArrayD, + dtype: &str, +) -> std::io::Result<()> { + match dtype { + "float32" => write_array_as_f32(writer, encoder_input), + "bfloat16" => write_array_as_u16(writer, encoder_input, f32_to_bf16_bits), + "float16" => write_array_as_u16(writer, encoder_input, f32_to_f16_bits), + _ => unreachable!("canonical dtype is constrained before direct SHM write"), + } +} + +fn write_array_as_f32(writer: &mut impl Write, encoder_input: &ArrayD) -> std::io::Result<()> { + if let Some(encoder_slice) = encoder_input + .as_slice() + .or_else(|| encoder_input.as_slice_memory_order()) + { + #[cfg(target_endian = "little")] + { + return writer.write_all(bytemuck::cast_slice(encoder_slice)); + } + #[cfg(not(target_endian = "little"))] + { + for value in encoder_slice { + writer.write_all(&value.to_le_bytes())?; + } + return Ok(()); + } + } + + for value in encoder_input.iter() { + writer.write_all(&value.to_le_bytes())?; + } + Ok(()) +} + +fn write_array_as_u16( + writer: &mut impl Write, + encoder_input: &ArrayD, + convert: F, +) -> std::io::Result<()> +where + F: Fn(f32) -> u16 + Copy, +{ + const CHUNK_VALUES: usize = 256 * 1024; + + let mut converted = Vec::with_capacity(CHUNK_VALUES); + let mut flush = |converted: &mut Vec| -> std::io::Result<()> { + if converted.is_empty() { + return Ok(()); + } + #[cfg(target_endian = "little")] + { + writer.write_all(bytemuck::cast_slice(converted.as_slice()))?; + } + #[cfg(not(target_endian = "little"))] + { + for value in converted.iter() { + writer.write_all(&value.to_le_bytes())?; + } + } + converted.clear(); + Ok(()) + }; + + if let Some(encoder_slice) = encoder_input.as_slice() { + for &value in encoder_slice { + converted.push(convert(value)); + if converted.len() == CHUNK_VALUES { + flush(&mut converted)?; + } + } + } else { + for &value in encoder_input.iter() { + converted.push(convert(value)); + if converted.len() == CHUNK_VALUES { + flush(&mut converted)?; + } + } + } + flush(&mut converted) +} + fn serialize_array_as_dtype( encoder_input: &ArrayD, dtype: &str, @@ -1276,18 +1493,12 @@ fn serialize_array_as_dtype( (data, shape, "float32".to_string()) } Some("bfloat16") => ( - encoder_input - .iter() - .flat_map(|value| f32_to_bf16_bits(*value).to_le_bytes()) - .collect(), + serialize_array_as_u16_bytes(encoder_input, f32_to_bf16_bits), array_shape(encoder_input), "bfloat16".to_string(), ), Some("float16") => ( - encoder_input - .iter() - .flat_map(|value| f32_to_f16_bits(*value).to_le_bytes()) - .collect(), + serialize_array_as_u16_bytes(encoder_input, f32_to_f16_bits), array_shape(encoder_input), "float16".to_string(), ), @@ -1302,6 +1513,33 @@ fn serialize_array_as_dtype( } } +fn serialize_array_as_u16_bytes(encoder_input: &ArrayD, convert: F) -> Vec +where + F: Fn(f32) -> u16 + Copy, +{ + let element_count = encoder_input.len(); + let mut converted = Vec::with_capacity(element_count); + + if let Some(encoder_slice) = encoder_input.as_slice() { + converted.extend(encoder_slice.iter().map(|&value| convert(value))); + } else { + converted.extend(encoder_input.iter().map(|&value| convert(value))); + } + + #[cfg(target_endian = "little")] + { + bytemuck::cast_slice(&converted).to_vec() + } + #[cfg(not(target_endian = "little"))] + { + let mut bytes = Vec::with_capacity(element_count * std::mem::size_of::()); + for value in converted { + bytes.extend_from_slice(&value.to_le_bytes()); + } + bytes + } +} + fn tokenspeed_encoder_input_dtype(modality: Modality, workers: Option<&WorkerSelection>) -> String { if let Some(dtype) = tokenspeed_encoder_input_dtype_from_env(modality) { return dtype; @@ -1365,6 +1603,7 @@ fn array_shape(encoder_input: &ArrayD) -> Vec { encoder_input.shape().iter().map(|&d| d as u32).collect() } +#[inline] fn f32_to_bf16_bits(value: f32) -> u16 { let bits = value.to_bits(); let lsb = (bits >> 16) & 1; @@ -1372,6 +1611,7 @@ fn f32_to_bf16_bits(value: f32) -> u16 { (bits.wrapping_add(rounding_bias) >> 16) as u16 } +#[inline] fn f32_to_f16_bits(value: f32) -> u16 { let bits = value.to_bits(); let sign = ((bits >> 16) & 0x8000) as u16; @@ -1761,8 +2001,8 @@ mod tests { let first = &assembled.items[0]; assert_eq!(first.modality, TokenSpeedModality::Image); - assert_eq!(first.encoder_input_shape, vec![2, 2]); - assert_eq!(first.encoder_input.len(), 4 * size_of::()); + assert_eq!(first.encoder_input.shape, vec![2, 2]); + assert_eq!(first.encoder_input.nbytes(), 4 * size_of::()); assert_eq!(first.mm_placeholders, vec![(10, 2)]); assert_eq!( first.content_hash, @@ -1778,7 +2018,7 @@ mod tests { ); let second = &assembled.items[1]; - assert_eq!(second.encoder_input_shape, vec![2, 2]); + assert_eq!(second.encoder_input.shape, vec![2, 2]); assert_eq!(second.mm_placeholders, vec![(20, 2)]); assert_eq!( second.content_hash, @@ -1867,8 +2107,8 @@ mod tests { let first = &assembled.items[0]; assert_eq!(first.modality, TokenSpeedModality::Video); - assert_eq!(first.encoder_input_shape, vec![2, 2]); - assert_eq!(first.encoder_input.len(), 4 * size_of::()); + assert_eq!(first.encoder_input.shape, vec![2, 2]); + assert_eq!(first.encoder_input.nbytes(), 4 * size_of::()); assert_eq!(first.mm_placeholders, vec![(30, 2)]); assert_eq!( first.content_hash, @@ -1884,7 +2124,7 @@ mod tests { ); let second = &assembled.items[1]; - assert_eq!(second.encoder_input_shape, vec![2, 2]); + assert_eq!(second.encoder_input.shape, vec![2, 2]); assert_eq!(second.mm_placeholders, vec![(40, 2)]); assert_eq!( second.content_hash, diff --git a/model_gateway/src/routers/grpc/proto_wrapper.rs b/model_gateway/src/routers/grpc/proto_wrapper.rs index 05960637d4..f3e45f38c1 100644 --- a/model_gateway/src/routers/grpc/proto_wrapper.rs +++ b/model_gateway/src/routers/grpc/proto_wrapper.rs @@ -4,7 +4,15 @@ //! supported backend, allowing the router to work with any backend //! transparently. -use std::collections::HashMap; +use std::{ + collections::HashMap, + fs::{remove_file, OpenOptions}, + io::Write, + path::PathBuf, + process, + sync::atomic::{AtomicU64, Ordering}, + time::{Instant, SystemTime, UNIX_EPOCH}, +}; use futures_util::StreamExt; use smg_grpc_client::{ @@ -75,7 +83,7 @@ pub struct TrtllmMultimodalData { pub image_data: Vec>, } -/// TokenSpeed multimodal data: preprocessed tensors with patch-only placeholders. +/// TokenSpeed multimodal data: preprocessed encoder input with patch-only placeholders. #[derive(Debug)] pub struct TokenSpeedMultimodalData { pub items: Vec, @@ -84,9 +92,7 @@ pub struct TokenSpeedMultimodalData { #[derive(Debug)] pub struct TokenSpeedMultimodalItem { pub modality: TokenSpeedModality, - pub encoder_input: Vec, - pub encoder_input_shape: Vec, - pub encoder_input_dtype: String, + pub encoder_input: TokenSpeedTensorBytes, pub model_specific_tensors: HashMap, pub placeholder_token_id: Option, pub mm_placeholders: Vec<(u32, u32)>, @@ -108,6 +114,47 @@ pub struct TensorBytes { pub dtype: String, } +/// TokenSpeed tensor bytes can either be materialized bytes or an already +/// published SHM handle. The latter lets SMG serialize large encoder inputs +/// directly into SHM instead of building a temporary Vec first. +#[derive(Debug, Clone)] +pub struct TokenSpeedTensorBytes { + pub payload: TokenSpeedTensorPayload, + pub shape: Vec, + pub dtype: String, +} + +#[derive(Debug, Clone)] +pub enum TokenSpeedTensorPayload { + Bytes(Vec), + Shm(tokenspeed::ShmHandle), +} + +impl TokenSpeedTensorBytes { + pub fn bytes(data: Vec, shape: Vec, dtype: String) -> Self { + Self { + payload: TokenSpeedTensorPayload::Bytes(data), + shape, + dtype, + } + } + + pub fn shm(handle: tokenspeed::ShmHandle, shape: Vec, dtype: String) -> Self { + Self { + payload: TokenSpeedTensorPayload::Shm(handle), + shape, + dtype, + } + } + + pub fn nbytes(&self) -> usize { + match &self.payload { + TokenSpeedTensorPayload::Bytes(data) => data.len(), + TokenSpeedTensorPayload::Shm(handle) => handle.nbytes as usize, + } + } +} + impl SglangMultimodalData { /// Convert to SGLang proto MultimodalInputs. pub fn into_proto(self) -> sglang::MultimodalInputs { @@ -235,11 +282,7 @@ impl TokenSpeedMultimodalItem { TokenSpeedModality::Video => tokenspeed::Modality::Video as i32, }, content_hash: self.content_hash, - encoder_input: Some(tensor_bytes_to_tokenspeed(TensorBytes { - data: self.encoder_input, - shape: self.encoder_input_shape, - dtype: self.encoder_input_dtype, - })), + encoder_input: Some(tokenspeed_tensor_bytes_to_proto(self.encoder_input)), model_specific_tensors, placeholders, placeholder_token_id: self.placeholder_token_id, @@ -247,15 +290,153 @@ impl TokenSpeedMultimodalItem { } } +fn tokenspeed_tensor_bytes_to_proto(value: TokenSpeedTensorBytes) -> tokenspeed::TensorData { + let TokenSpeedTensorBytes { + payload, + shape, + dtype, + } = value; + let payload = match payload { + TokenSpeedTensorPayload::Bytes(data) => tokenspeed_tensor_payload(data), + TokenSpeedTensorPayload::Shm(handle) => tokenspeed::tensor_data::Payload::Shm(handle), + }; + + tokenspeed::TensorData { + shape, + dtype, + payload: Some(payload), + } +} + fn tensor_bytes_to_tokenspeed(value: TensorBytes) -> tokenspeed::TensorData { - let data = value.data; + let TensorBytes { data, shape, dtype } = value; + tokenspeed::TensorData { - shape: value.shape, - dtype: value.dtype, - payload: Some(tokenspeed::tensor_data::Payload::Inline(data)), + shape, + dtype, + payload: Some(tokenspeed_tensor_payload(data)), + } +} + +fn tokenspeed_tensor_payload(data: Vec) -> tokenspeed::tensor_data::Payload { + let log_timing = log_tokenspeed_mm_timing_enabled(); + let mode = std::env::var("SMG_TOKENSPEED_TENSOR_TRANSPORT") + .unwrap_or_default() + .trim() + .to_ascii_lowercase(); + if mode != "shm" { + if log_timing { + tracing::info!( + nbytes = data.len(), + "smg_mm_timing tokenspeed_tensor_payload_inline" + ); + } + return tokenspeed::tensor_data::Payload::Inline(data); + } + + let min_bytes = std::env::var("SMG_TOKENSPEED_SHM_MIN_BYTES") + .ok() + .and_then(|value| value.parse::().ok()) + .unwrap_or(64 * 1024); + if data.len() < min_bytes { + if log_timing { + tracing::info!( + nbytes = data.len(), + min_bytes, + "smg_mm_timing tokenspeed_tensor_payload_inline_below_threshold" + ); + } + return tokenspeed::tensor_data::Payload::Inline(data); + } + + let started = Instant::now(); + match write_tokenspeed_shm(&data) { + Ok(handle) => { + if log_timing { + tracing::info!( + nbytes = data.len(), + elapsed_ms = started.elapsed().as_secs_f64() * 1000.0, + "smg_mm_timing tokenspeed_shm_write" + ); + } + tokenspeed::tensor_data::Payload::Shm(handle) + } + Err(error) => { + tracing::warn!( + ?error, + nbytes = data.len(), + "Failed to create TokenSpeed SHM tensor payload; falling back to inline" + ); + tokenspeed::tensor_data::Payload::Inline(data) + } } } +fn log_tokenspeed_mm_timing_enabled() -> bool { + std::env::var("SMG_LOG_MM_TIMING") + .map(|value| matches!(value.to_ascii_lowercase().as_str(), "1" | "true" | "yes")) + .unwrap_or(false) +} + +pub fn tokenspeed_shm_transport_enabled_for_bytes(nbytes: usize) -> bool { + let mode = std::env::var("SMG_TOKENSPEED_TENSOR_TRANSPORT") + .unwrap_or_default() + .trim() + .to_ascii_lowercase(); + if mode != "shm" { + return false; + } + + let min_bytes = std::env::var("SMG_TOKENSPEED_SHM_MIN_BYTES") + .ok() + .and_then(|value| value.parse::().ok()) + .unwrap_or(64 * 1024); + nbytes >= min_bytes +} + +static TOKENSPEED_SHM_COUNTER: AtomicU64 = AtomicU64::new(0); + +fn write_tokenspeed_shm(data: &[u8]) -> std::io::Result { + write_tokenspeed_shm_with(data.len(), |file| file.write_all(data)) +} + +pub fn write_tokenspeed_shm_with( + nbytes: usize, + write_fn: impl FnOnce(&mut std::fs::File) -> std::io::Result<()>, +) -> std::io::Result { + let name = next_tokenspeed_shm_name(); + let path = tokenspeed_shm_path(&name); + let mut file = OpenOptions::new() + .write(true) + .create_new(true) + .open(&path)?; + if let Err(error) = write_fn(&mut file) { + drop(file); + let _ = remove_file(&path); + return Err(error); + } + + Ok(tokenspeed::ShmHandle { + name, + offset: 0, + nbytes: nbytes as u64, + owner_id: format!("smg:{}", process::id()), + }) +} + +fn next_tokenspeed_shm_name() -> String { + let seq = TOKENSPEED_SHM_COUNTER.fetch_add(1, Ordering::Relaxed); + let nanos = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map(|duration| duration.as_nanos()) + .unwrap_or_default(); + format!("smg-tokenspeed-{}-{}-{}", process::id(), nanos, seq) +} + +fn tokenspeed_shm_path(name: &str) -> PathBuf { + PathBuf::from("/dev/shm").join(name) +} + // ===================== // Unified Logprobs Types // ===================== @@ -1307,9 +1488,11 @@ mod tests { let proto = TokenSpeedMultimodalData { items: vec![TokenSpeedMultimodalItem { modality: TokenSpeedModality::Image, - encoder_input: vec![42; 8], - encoder_input_shape: vec![1, 2], - encoder_input_dtype: "float32".to_string(), + encoder_input: TokenSpeedTensorBytes::bytes( + vec![42; 8], + vec![1, 2], + "float32".to_string(), + ), model_specific_tensors, placeholder_token_id: Some(151655), mm_placeholders: vec![(4, 2)], @@ -1346,9 +1529,11 @@ mod tests { let proto = TokenSpeedMultimodalData { items: vec![TokenSpeedMultimodalItem { modality: TokenSpeedModality::Video, - encoder_input: vec![42; 8], - encoder_input_shape: vec![1, 2], - encoder_input_dtype: "float32".to_string(), + encoder_input: TokenSpeedTensorBytes::bytes( + vec![42; 8], + vec![1, 2], + "float32".to_string(), + ), model_specific_tensors, placeholder_token_id: Some(151656), mm_placeholders: vec![(4, 2)], @@ -1388,6 +1573,41 @@ mod tests { ); } + #[test] + fn tokenspeed_shm_encoder_input_into_proto_uses_shm_payload() { + let proto = TokenSpeedMultimodalData { + items: vec![TokenSpeedMultimodalItem { + modality: TokenSpeedModality::Image, + encoder_input: TokenSpeedTensorBytes::shm( + tokenspeed::ShmHandle { + name: "smg-test-shm".to_string(), + offset: 0, + nbytes: 8, + owner_id: "smg:test".to_string(), + }, + vec![1, 2], + "bfloat16".to_string(), + ), + model_specific_tensors: HashMap::new(), + placeholder_token_id: Some(151655), + mm_placeholders: vec![(4, 2)], + content_hash: vec![7; 32], + }], + } + .into_proto(); + + let tensor = proto.items[0].encoder_input.as_ref().unwrap(); + assert_eq!(tensor.shape, vec![1, 2]); + assert_eq!(tensor.dtype, "bfloat16"); + match tensor.payload.as_ref() { + Some(tokenspeed::tensor_data::Payload::Shm(handle)) => { + assert_eq!(handle.name, "smg-test-shm"); + assert_eq!(handle.nbytes, 8); + } + _ => panic!("expected shm TensorData payload"), + } + } + fn inline_tensor_data(tensor: &tokenspeed::TensorData) -> &[u8] { match tensor.payload.as_ref() { Some(tokenspeed::tensor_data::Payload::Inline(data)) => data, From 0dac842c2e843a143dddabedde4cfe212251cbf7 Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Thu, 4 Jun 2026 05:50:49 -0700 Subject: [PATCH 02/28] Tighten multimodal transport validation Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- model_gateway/src/routers/grpc/proto_wrapper.rs | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/model_gateway/src/routers/grpc/proto_wrapper.rs b/model_gateway/src/routers/grpc/proto_wrapper.rs index f3e45f38c1..bcc42c6190 100644 --- a/model_gateway/src/routers/grpc/proto_wrapper.rs +++ b/model_gateway/src/routers/grpc/proto_wrapper.rs @@ -415,6 +415,19 @@ pub fn write_tokenspeed_shm_with( let _ = remove_file(&path); return Err(error); } + if let Err(error) = file.flush().and_then(|_| { + if file.metadata()?.len() != nbytes as u64 { + return Err(std::io::Error::new( + std::io::ErrorKind::InvalidData, + "TokenSpeed SHM writer produced an unexpected byte length", + )); + } + Ok(()) + }) { + drop(file); + let _ = remove_file(&path); + return Err(error); + } Ok(tokenspeed::ShmHandle { name, From 663b820313d86cbcfd3e1b01b058128670ec12df Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Thu, 4 Jun 2026 05:59:08 -0700 Subject: [PATCH 03/28] Rename TokenSpeed tensor storage wrapper Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- model_gateway/src/routers/grpc/multimodal.rs | 8 +-- .../src/routers/grpc/proto_wrapper.rs | 50 +++++++++---------- 2 files changed, 29 insertions(+), 29 deletions(-) diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index 0a2eb88f48..8780260547 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -41,7 +41,7 @@ use crate::routers::grpc::{ proto_wrapper::{ tokenspeed_shm_transport_enabled_for_bytes, write_tokenspeed_shm_with, SglangMultimodalData, TensorBytes, TokenSpeedModality, TokenSpeedMultimodalData, - TokenSpeedMultimodalItem, TokenSpeedTensorBytes, TrtllmMultimodalData, VllmMultimodalData, + TokenSpeedMultimodalItem, TokenSpeedTensor, TrtllmMultimodalData, VllmMultimodalData, }, MultimodalData, }; @@ -1348,7 +1348,7 @@ fn serialize_array(encoder_input: &ArrayD) -> (Vec, Vec) { fn serialize_array_as_tokenspeed_tensor( encoder_input: &ArrayD, dtype: &str, -) -> TokenSpeedTensorBytes { +) -> TokenSpeedTensor { let dtype = match canonical_float_dtype(dtype).as_deref() { Some("float32") => "float32".to_string(), Some("bfloat16") => "bfloat16".to_string(), @@ -1382,7 +1382,7 @@ fn serialize_array_as_tokenspeed_tensor( "smg_mm_timing tokenspeed_shm_write_direct" ); } - return TokenSpeedTensorBytes::shm(handle, shape, dtype); + return TokenSpeedTensor::shm(handle, shape, dtype); } Err(error) => { warn!( @@ -1396,7 +1396,7 @@ fn serialize_array_as_tokenspeed_tensor( } let (data, shape, dtype) = serialize_array_as_dtype(encoder_input, &dtype); - TokenSpeedTensorBytes::bytes(data, shape, dtype) + TokenSpeedTensor::inline(data, shape, dtype) } fn write_array_as_dtype( diff --git a/model_gateway/src/routers/grpc/proto_wrapper.rs b/model_gateway/src/routers/grpc/proto_wrapper.rs index bcc42c6190..f0e43890d6 100644 --- a/model_gateway/src/routers/grpc/proto_wrapper.rs +++ b/model_gateway/src/routers/grpc/proto_wrapper.rs @@ -92,7 +92,7 @@ pub struct TokenSpeedMultimodalData { #[derive(Debug)] pub struct TokenSpeedMultimodalItem { pub modality: TokenSpeedModality, - pub encoder_input: TokenSpeedTensorBytes, + pub encoder_input: TokenSpeedTensor, pub model_specific_tensors: HashMap, pub placeholder_token_id: Option, pub mm_placeholders: Vec<(u32, u32)>, @@ -114,26 +114,26 @@ pub struct TensorBytes { pub dtype: String, } -/// TokenSpeed tensor bytes can either be materialized bytes or an already -/// published SHM handle. The latter lets SMG serialize large encoder inputs -/// directly into SHM instead of building a temporary Vec first. +/// TokenSpeed tensor with storage that can be either inline bytes or an +/// already-published SHM handle. The latter lets SMG serialize large encoder +/// inputs directly into SHM instead of building a temporary Vec first. #[derive(Debug, Clone)] -pub struct TokenSpeedTensorBytes { - pub payload: TokenSpeedTensorPayload, +pub struct TokenSpeedTensor { + pub storage: TokenSpeedTensorStorage, pub shape: Vec, pub dtype: String, } #[derive(Debug, Clone)] -pub enum TokenSpeedTensorPayload { - Bytes(Vec), +pub enum TokenSpeedTensorStorage { + Inline(Vec), Shm(tokenspeed::ShmHandle), } -impl TokenSpeedTensorBytes { - pub fn bytes(data: Vec, shape: Vec, dtype: String) -> Self { +impl TokenSpeedTensor { + pub fn inline(data: Vec, shape: Vec, dtype: String) -> Self { Self { - payload: TokenSpeedTensorPayload::Bytes(data), + storage: TokenSpeedTensorStorage::Inline(data), shape, dtype, } @@ -141,16 +141,16 @@ impl TokenSpeedTensorBytes { pub fn shm(handle: tokenspeed::ShmHandle, shape: Vec, dtype: String) -> Self { Self { - payload: TokenSpeedTensorPayload::Shm(handle), + storage: TokenSpeedTensorStorage::Shm(handle), shape, dtype, } } pub fn nbytes(&self) -> usize { - match &self.payload { - TokenSpeedTensorPayload::Bytes(data) => data.len(), - TokenSpeedTensorPayload::Shm(handle) => handle.nbytes as usize, + match &self.storage { + TokenSpeedTensorStorage::Inline(data) => data.len(), + TokenSpeedTensorStorage::Shm(handle) => handle.nbytes as usize, } } } @@ -282,7 +282,7 @@ impl TokenSpeedMultimodalItem { TokenSpeedModality::Video => tokenspeed::Modality::Video as i32, }, content_hash: self.content_hash, - encoder_input: Some(tokenspeed_tensor_bytes_to_proto(self.encoder_input)), + encoder_input: Some(tokenspeed_tensor_to_proto(self.encoder_input)), model_specific_tensors, placeholders, placeholder_token_id: self.placeholder_token_id, @@ -290,15 +290,15 @@ impl TokenSpeedMultimodalItem { } } -fn tokenspeed_tensor_bytes_to_proto(value: TokenSpeedTensorBytes) -> tokenspeed::TensorData { - let TokenSpeedTensorBytes { - payload, +fn tokenspeed_tensor_to_proto(value: TokenSpeedTensor) -> tokenspeed::TensorData { + let TokenSpeedTensor { + storage, shape, dtype, } = value; - let payload = match payload { - TokenSpeedTensorPayload::Bytes(data) => tokenspeed_tensor_payload(data), - TokenSpeedTensorPayload::Shm(handle) => tokenspeed::tensor_data::Payload::Shm(handle), + let payload = match storage { + TokenSpeedTensorStorage::Inline(data) => tokenspeed_tensor_payload(data), + TokenSpeedTensorStorage::Shm(handle) => tokenspeed::tensor_data::Payload::Shm(handle), }; tokenspeed::TensorData { @@ -1501,7 +1501,7 @@ mod tests { let proto = TokenSpeedMultimodalData { items: vec![TokenSpeedMultimodalItem { modality: TokenSpeedModality::Image, - encoder_input: TokenSpeedTensorBytes::bytes( + encoder_input: TokenSpeedTensor::inline( vec![42; 8], vec![1, 2], "float32".to_string(), @@ -1542,7 +1542,7 @@ mod tests { let proto = TokenSpeedMultimodalData { items: vec![TokenSpeedMultimodalItem { modality: TokenSpeedModality::Video, - encoder_input: TokenSpeedTensorBytes::bytes( + encoder_input: TokenSpeedTensor::inline( vec![42; 8], vec![1, 2], "float32".to_string(), @@ -1591,7 +1591,7 @@ mod tests { let proto = TokenSpeedMultimodalData { items: vec![TokenSpeedMultimodalItem { modality: TokenSpeedModality::Image, - encoder_input: TokenSpeedTensorBytes::shm( + encoder_input: TokenSpeedTensor::shm( tokenspeed::ShmHandle { name: "smg-test-shm".to_string(), offset: 0, From c3c87619b8ae7ca8abf329cb758cca10e24ceddc Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Thu, 4 Jun 2026 19:30:48 -0700 Subject: [PATCH 04/28] Clean up TokenSpeed SHM payloads on send failures Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- model_gateway/src/routers/grpc/client.rs | 44 +++++++++-- .../src/routers/grpc/proto_wrapper.rs | 75 +++++++++++++++++++ 2 files changed, 112 insertions(+), 7 deletions(-) diff --git a/model_gateway/src/routers/grpc/client.rs b/model_gateway/src/routers/grpc/client.rs index 436dd13ca2..fae781a141 100644 --- a/model_gateway/src/routers/grpc/client.rs +++ b/model_gateway/src/routers/grpc/client.rs @@ -13,7 +13,11 @@ use smg_grpc_client::{ }; use crate::routers::grpc::{ - proto_wrapper::{ProtoEmbedComplete, ProtoEmbedRequest, ProtoGenerateRequest, ProtoStream}, + proto_wrapper::{ + cleanup_tokenspeed_shm_handles, collect_tokenspeed_generate_request_shm_handles, + collect_tokenspeed_multimodal_inputs_shm_handles, ProtoEmbedComplete, ProtoEmbedRequest, + ProtoGenerateRequest, ProtoStream, + }, MultimodalData, }; @@ -405,8 +409,14 @@ impl GrpcClient { Ok(ProtoStream::Mlx(stream)) } (Self::TokenSpeed(client), ProtoGenerateRequest::TokenSpeed(boxed_req)) => { - let stream = client.generate(*boxed_req).await?; - Ok(ProtoStream::TokenSpeed(stream)) + let shm_handles = collect_tokenspeed_generate_request_shm_handles(&boxed_req); + match client.generate(*boxed_req).await { + Ok(stream) => Ok(ProtoStream::TokenSpeed(stream)), + Err(error) => { + cleanup_tokenspeed_shm_handles(&shm_handles); + Err(error) + } + } } #[expect( clippy::panic, @@ -520,14 +530,24 @@ impl GrpcClient { MultimodalData::TokenSpeed(data) => data.into_proto(), _ => unreachable!("caller guarantees matching variant"), }); - let req = client.build_generate_request_from_chat( + let shm_handles = tokenspeed_mm + .as_ref() + .map(collect_tokenspeed_multimodal_inputs_shm_handles) + .unwrap_or_default(); + let req = match client.build_generate_request_from_chat( request_id, body, processed_text, token_ids, tokenspeed_mm, options.tool_constraints, - )?; + ) { + Ok(req) => req, + Err(error) => { + cleanup_tokenspeed_shm_handles(&shm_handles); + return Err(error); + } + }; Ok(ProtoGenerateRequest::TokenSpeed(Box::new(req))) } } @@ -610,14 +630,24 @@ impl GrpcClient { MultimodalData::TokenSpeed(data) => data.into_proto(), _ => unreachable!("caller guarantees matching variant"), }); - let req = client.build_generate_request_from_messages( + let shm_handles = tokenspeed_mm + .as_ref() + .map(collect_tokenspeed_multimodal_inputs_shm_handles) + .unwrap_or_default(); + let req = match client.build_generate_request_from_messages( request_id, body, processed_text, token_ids, tokenspeed_mm, options.tool_constraints, - )?; + ) { + Ok(req) => req, + Err(error) => { + cleanup_tokenspeed_shm_handles(&shm_handles); + return Err(error); + } + }; Ok(ProtoGenerateRequest::TokenSpeed(Box::new(req))) } } diff --git a/model_gateway/src/routers/grpc/proto_wrapper.rs b/model_gateway/src/routers/grpc/proto_wrapper.rs index f0e43890d6..cbecd7b5a0 100644 --- a/model_gateway/src/routers/grpc/proto_wrapper.rs +++ b/model_gateway/src/routers/grpc/proto_wrapper.rs @@ -437,6 +437,81 @@ pub fn write_tokenspeed_shm_with( }) } +pub fn collect_tokenspeed_multimodal_inputs_shm_handles( + inputs: &tokenspeed::MultimodalInputs, +) -> Vec { + let mut handles = Vec::new(); + for item in &inputs.items { + collect_optional_tokenspeed_tensor_shm_handles(&item.encoder_input, &mut handles); + for tensor in item.model_specific_tensors.values() { + collect_tokenspeed_tensor_shm_handles(tensor, &mut handles); + } + } + handles +} + +pub fn collect_tokenspeed_generate_request_shm_handles( + request: &tokenspeed::GenerateRequest, +) -> Vec { + request + .mm_inputs + .as_ref() + .map(collect_tokenspeed_multimodal_inputs_shm_handles) + .unwrap_or_default() +} + +pub fn cleanup_tokenspeed_shm_handles(handles: &[tokenspeed::ShmHandle]) { + for handle in handles { + let Some(name) = validate_tokenspeed_shm_name_for_cleanup(&handle.name) else { + tracing::warn!( + name = %handle.name, + "Skipping cleanup for invalid TokenSpeed SHM name" + ); + continue; + }; + let path = tokenspeed_shm_path(name); + match remove_file(&path) { + Ok(()) => {} + Err(error) if error.kind() == std::io::ErrorKind::NotFound => {} + Err(error) => { + tracing::warn!( + ?error, + path = %path.display(), + "Failed to cleanup TokenSpeed SHM file" + ); + } + } + } +} + +fn collect_optional_tokenspeed_tensor_shm_handles( + tensor: &Option, + handles: &mut Vec, +) { + let Some(tensor) = tensor else { + return; + }; + collect_tokenspeed_tensor_shm_handles(tensor, handles); +} + +fn collect_tokenspeed_tensor_shm_handles( + tensor: &tokenspeed::TensorData, + handles: &mut Vec, +) { + if let Some(tokenspeed::tensor_data::Payload::Shm(handle)) = &tensor.payload { + handles.push(handle.clone()); + } +} + +fn validate_tokenspeed_shm_name_for_cleanup(name: &str) -> Option<&str> { + let name = name.strip_prefix('/').unwrap_or(name); + if name.is_empty() || name.contains('/') || name == "." || name == ".." || name.contains('\0') + { + return None; + } + Some(name) +} + fn next_tokenspeed_shm_name() -> String { let seq = TOKENSPEED_SHM_COUNTER.fetch_add(1, Ordering::Relaxed); let nanos = SystemTime::now() From 4876fa98d71f1fe80af91507456015390adb67bb Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Tue, 9 Jun 2026 08:33:33 -0700 Subject: [PATCH 05/28] style(tokenspeed): format shm servicer path Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- .../smg_grpc_servicer/tokenspeed/servicer.py | 46 +++++++------------ 1 file changed, 17 insertions(+), 29 deletions(-) diff --git a/grpc_servicer/smg_grpc_servicer/tokenspeed/servicer.py b/grpc_servicer/smg_grpc_servicer/tokenspeed/servicer.py index 91fcd1f2ac..be2064c640 100644 --- a/grpc_servicer/smg_grpc_servicer/tokenspeed/servicer.py +++ b/grpc_servicer/smg_grpc_servicer/tokenspeed/servicer.py @@ -63,9 +63,11 @@ "true", "yes", ) -UNLINK_MM_SHM_AFTER_READ = os.getenv( - "TOKENSPEED_UNLINK_MM_SHM_AFTER_READ", "1" -).lower() not in ("0", "false", "no") +UNLINK_MM_SHM_AFTER_READ = os.getenv("TOKENSPEED_UNLINK_MM_SHM_AFTER_READ", "1").lower() not in ( + "0", + "false", + "no", +) def _lazy_generate_req_input(): @@ -153,9 +155,7 @@ async def Generate( build_started = time.perf_counter() req_obj = self._build_generate_req(request) if LOG_MM_TIMING: - has_mm = ( - getattr(req_obj, "precomputed_multimodal_inputs", None) is not None - ) + has_mm = getattr(req_obj, "precomputed_multimodal_inputs", None) is not None logger.info( "mm_timing generate_build_ms rid=%s elapsed=%.3f has_mm=%s", rid, @@ -961,9 +961,7 @@ def _mm_inputs_from_itemized_proto( raise ValueError("MultimodalItem must include encoder_input") feature_started = time.perf_counter() if LOG_MM_TIMING else None - feature = self._feature_from_proto( - item_proto.encoder_input, cast_to=model_dtype - ) + feature = self._feature_from_proto(item_proto.encoder_input, cast_to=model_dtype) feature_elapsed_ms = ( (time.perf_counter() - feature_started) * 1000 if feature_started is not None @@ -992,9 +990,7 @@ def _mm_inputs_from_itemized_proto( for name, tensor_data in item_proto.model_specific_tensors.items() } model_elapsed_ms = ( - (time.perf_counter() - model_started) * 1000 - if model_started is not None - else None + (time.perf_counter() - model_started) * 1000 if model_started is not None else None ) self._validate_item_tensor_consistency(modality, model_specific_data) @@ -1143,28 +1139,22 @@ def _feature_from_proto( materializing bytes and cloning a CPU tensor. """ if tensor_data.WhichOneof("payload") != "shm": - return TokenSpeedSchedulerServicer._tensor_from_proto( - tensor_data, cast_to=cast_to - ) + return TokenSpeedSchedulerServicer._tensor_from_proto(tensor_data, cast_to=cast_to) dtype = TokenSpeedSchedulerServicer._torch_dtype_from_proto(tensor_data.dtype) - if cast_to is not None and dtype != cast_to and torch.is_floating_point( - torch.empty((), dtype=dtype) + if ( + cast_to is not None + and dtype != cast_to + and torch.is_floating_point(torch.empty((), dtype=dtype)) ): - return TokenSpeedSchedulerServicer._tensor_from_proto( - tensor_data, cast_to=cast_to - ) + return TokenSpeedSchedulerServicer._tensor_from_proto(tensor_data, cast_to=cast_to) shm = tensor_data.shm if shm.offset != 0: - return TokenSpeedSchedulerServicer._tensor_from_proto( - tensor_data, cast_to=cast_to - ) + return TokenSpeedSchedulerServicer._tensor_from_proto(tensor_data, cast_to=cast_to) shape = tuple(int(dim) for dim in tensor_data.shape) - expected = int(np.prod(shape, dtype=np.int64)) * torch.empty( - (), dtype=dtype - ).element_size() + expected = int(np.prod(shape, dtype=np.int64)) * torch.empty((), dtype=dtype).element_size() if int(shm.nbytes) != expected: raise ValueError( f"TensorData.shm byte length mismatch for dtype={tensor_data.dtype}, " @@ -1180,9 +1170,7 @@ def _tensor_payload_bytes(tensor_data: tokenspeed_scheduler_pb2.TensorData) -> b if payload == "inline": return bytes(tensor_data.inline) if payload == "shm": - return TokenSpeedSchedulerServicer._tensor_payload_bytes_from_shm( - tensor_data.shm - ) + return TokenSpeedSchedulerServicer._tensor_payload_bytes_from_shm(tensor_data.shm) if payload == "remote": raise ValueError("TensorData.remote payload is not implemented yet") raise ValueError("TensorData payload is required") From 0bdfdb5a53b28119423d8e0e736f2cc5839e9c6d Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Tue, 9 Jun 2026 08:47:18 -0700 Subject: [PATCH 06/28] fix(ci): satisfy clippy for multimodal transport Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- model_gateway/src/routers/grpc/multimodal.rs | 27 ++++++++++--------- .../src/routers/grpc/proto_wrapper.rs | 9 +++---- 2 files changed, 19 insertions(+), 17 deletions(-) diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index 8780260547..554ad6b0b3 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -798,12 +798,12 @@ async fn process_multimodal_parts( let image_count = images.len(); let video_count = videos.len(); let video_frame_count = videos.first().map_or(0, |video| { - if !video.frames().is_empty() { - video.frames().len() - } else { + if video.frames().is_empty() { video .rgb_video() .map_or(0, |rgb_video| rgb_video.frames.len()) + } else { + video.frames().len() } }); let original_tokens = token_ids.len(); @@ -1362,12 +1362,12 @@ fn serialize_array_as_tokenspeed_tensor( } }; let shape = array_shape(encoder_input); - let nbytes = encoder_input.len() - * match dtype.as_str() { - "float32" => size_of::(), - "bfloat16" | "float16" => size_of::(), - _ => unreachable!("canonical dtype is constrained above"), - }; + let element_size = if dtype == "bfloat16" || dtype == "float16" { + size_of::() + } else { + size_of::() + }; + let nbytes = encoder_input.len() * element_size; if tokenspeed_shm_transport_enabled_for_bytes(nbytes) { let started = Instant::now(); @@ -1408,7 +1408,10 @@ fn write_array_as_dtype( "float32" => write_array_as_f32(writer, encoder_input), "bfloat16" => write_array_as_u16(writer, encoder_input, f32_to_bf16_bits), "float16" => write_array_as_u16(writer, encoder_input, f32_to_f16_bits), - _ => unreachable!("canonical dtype is constrained before direct SHM write"), + other => Err(std::io::Error::new( + std::io::ErrorKind::InvalidInput, + format!("unsupported TokenSpeed encoder input dtype: {other}"), + )), } } @@ -1430,7 +1433,7 @@ fn write_array_as_f32(writer: &mut impl Write, encoder_input: &ArrayD) -> s } } - for value in encoder_input.iter() { + for value in encoder_input { writer.write_all(&value.to_le_bytes())?; } Ok(()) @@ -1473,7 +1476,7 @@ where } } } else { - for &value in encoder_input.iter() { + for &value in encoder_input { converted.push(convert(value)); if converted.len() == CHUNK_VALUES { flush(&mut converted)?; diff --git a/model_gateway/src/routers/grpc/proto_wrapper.rs b/model_gateway/src/routers/grpc/proto_wrapper.rs index cbecd7b5a0..dc533af0f7 100644 --- a/model_gateway/src/routers/grpc/proto_wrapper.rs +++ b/model_gateway/src/routers/grpc/proto_wrapper.rs @@ -415,7 +415,7 @@ pub fn write_tokenspeed_shm_with( let _ = remove_file(&path); return Err(error); } - if let Err(error) = file.flush().and_then(|_| { + if let Err(error) = file.flush().and_then(|()| { if file.metadata()?.len() != nbytes as u64 { return Err(std::io::Error::new( std::io::ErrorKind::InvalidData, @@ -442,7 +442,7 @@ pub fn collect_tokenspeed_multimodal_inputs_shm_handles( ) -> Vec { let mut handles = Vec::new(); for item in &inputs.items { - collect_optional_tokenspeed_tensor_shm_handles(&item.encoder_input, &mut handles); + collect_optional_tokenspeed_tensor_shm_handles(item.encoder_input.as_ref(), &mut handles); for tensor in item.model_specific_tensors.values() { collect_tokenspeed_tensor_shm_handles(tensor, &mut handles); } @@ -485,7 +485,7 @@ pub fn cleanup_tokenspeed_shm_handles(handles: &[tokenspeed::ShmHandle]) { } fn collect_optional_tokenspeed_tensor_shm_handles( - tensor: &Option, + tensor: Option<&tokenspeed::TensorData>, handles: &mut Vec, ) { let Some(tensor) = tensor else { @@ -505,8 +505,7 @@ fn collect_tokenspeed_tensor_shm_handles( fn validate_tokenspeed_shm_name_for_cleanup(name: &str) -> Option<&str> { let name = name.strip_prefix('/').unwrap_or(name); - if name.is_empty() || name.contains('/') || name == "." || name == ".." || name.contains('\0') - { + if name.is_empty() || name.contains('/') || name == "." || name == ".." || name.contains('\0') { return None; } Some(name) From 1d5e30fad3a966176779c4111971512a339061fe Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Mon, 15 Jun 2026 23:48:58 -0700 Subject: [PATCH 07/28] perf(multimodal): optimize TokenSpeed SHM serialization Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- .../proto/tokenspeed_scheduler.proto | 6 +- model_gateway/src/routers/grpc/multimodal.rs | 66 ++++++++++++------- .../src/routers/grpc/proto_wrapper.rs | 17 ++--- 3 files changed, 54 insertions(+), 35 deletions(-) diff --git a/crates/grpc_client/proto/tokenspeed_scheduler.proto b/crates/grpc_client/proto/tokenspeed_scheduler.proto index 5d104ce807..515143b441 100644 --- a/crates/grpc_client/proto/tokenspeed_scheduler.proto +++ b/crates/grpc_client/proto/tokenspeed_scheduler.proto @@ -123,9 +123,11 @@ message TensorData { oneof payload { // Current path: raw little-endian bytes carried in the gRPC message. bytes inline = 3; - // Same-host CPU shared memory path. + // Same-host CPU shared memory path. This is the preferred large-payload + // transport when SMG and TokenSpeed share /dev/shm. ShmHandle shm = 4; - // Transport-specific remote descriptor (NIXL/RDMA/object-store/etc.). + // Cross-node or non-shared-memory transport descriptor. NIXL is the + // expected remote transport for distributed multimodal tensor payloads. RemoteTensorHandle remote = 5; } } diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index 554ad6b0b3..ef5e46d185 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -1420,17 +1420,7 @@ fn write_array_as_f32(writer: &mut impl Write, encoder_input: &ArrayD) -> s .as_slice() .or_else(|| encoder_input.as_slice_memory_order()) { - #[cfg(target_endian = "little")] - { - return writer.write_all(bytemuck::cast_slice(encoder_slice)); - } - #[cfg(not(target_endian = "little"))] - { - for value in encoder_slice { - writer.write_all(&value.to_le_bytes())?; - } - return Ok(()); - } + return write_f32_slice(writer, encoder_slice); } for value in encoder_input { @@ -1439,6 +1429,20 @@ fn write_array_as_f32(writer: &mut impl Write, encoder_input: &ArrayD) -> s Ok(()) } +fn write_f32_slice(writer: &mut impl Write, values: &[f32]) -> std::io::Result<()> { + #[cfg(target_endian = "little")] + { + writer.write_all(bytemuck::cast_slice(values)) + } + #[cfg(not(target_endian = "little"))] + { + for value in values { + writer.write_all(&value.to_le_bytes())?; + } + Ok(()) + } +} + fn write_array_as_u16( writer: &mut impl Write, encoder_input: &ArrayD, @@ -1447,6 +1451,24 @@ fn write_array_as_u16( where F: Fn(f32) -> u16 + Copy, { + if let Some(encoder_slice) = encoder_input + .as_slice() + .or_else(|| encoder_input.as_slice_memory_order()) + { + let converted: Vec = encoder_slice.iter().map(|&value| convert(value)).collect(); + #[cfg(target_endian = "little")] + { + return writer.write_all(bytemuck::cast_slice(converted.as_slice())); + } + #[cfg(not(target_endian = "little"))] + { + for value in converted { + writer.write_all(&value.to_le_bytes())?; + } + return Ok(()); + } + } + const CHUNK_VALUES: usize = 256 * 1024; let mut converted = Vec::with_capacity(CHUNK_VALUES); @@ -1468,19 +1490,10 @@ where Ok(()) }; - if let Some(encoder_slice) = encoder_input.as_slice() { - for &value in encoder_slice { - converted.push(convert(value)); - if converted.len() == CHUNK_VALUES { - flush(&mut converted)?; - } - } - } else { - for &value in encoder_input { - converted.push(convert(value)); - if converted.len() == CHUNK_VALUES { - flush(&mut converted)?; - } + for &value in encoder_input { + converted.push(convert(value)); + if converted.len() == CHUNK_VALUES { + flush(&mut converted)?; } } flush(&mut converted) @@ -1523,7 +1536,10 @@ where let element_count = encoder_input.len(); let mut converted = Vec::with_capacity(element_count); - if let Some(encoder_slice) = encoder_input.as_slice() { + if let Some(encoder_slice) = encoder_input + .as_slice() + .or_else(|| encoder_input.as_slice_memory_order()) + { converted.extend(encoder_slice.iter().map(|&value| convert(value))); } else { converted.extend(encoder_input.iter().map(|&value| convert(value))); diff --git a/model_gateway/src/routers/grpc/proto_wrapper.rs b/model_gateway/src/routers/grpc/proto_wrapper.rs index dc533af0f7..56be720501 100644 --- a/model_gateway/src/routers/grpc/proto_wrapper.rs +++ b/model_gateway/src/routers/grpc/proto_wrapper.rs @@ -7,7 +7,7 @@ use std::{ collections::HashMap, fs::{remove_file, OpenOptions}, - io::Write, + io::{BufWriter, Write}, path::PathBuf, process, sync::atomic::{AtomicU64, Ordering}, @@ -402,21 +402,22 @@ fn write_tokenspeed_shm(data: &[u8]) -> std::io::Result { pub fn write_tokenspeed_shm_with( nbytes: usize, - write_fn: impl FnOnce(&mut std::fs::File) -> std::io::Result<()>, + write_fn: impl FnOnce(&mut BufWriter) -> std::io::Result<()>, ) -> std::io::Result { let name = next_tokenspeed_shm_name(); let path = tokenspeed_shm_path(&name); - let mut file = OpenOptions::new() + let file = OpenOptions::new() .write(true) .create_new(true) .open(&path)?; - if let Err(error) = write_fn(&mut file) { - drop(file); + let mut writer = BufWriter::new(file); + if let Err(error) = write_fn(&mut writer) { + drop(writer); let _ = remove_file(&path); return Err(error); } - if let Err(error) = file.flush().and_then(|()| { - if file.metadata()?.len() != nbytes as u64 { + if let Err(error) = writer.flush().and_then(|()| { + if writer.get_ref().metadata()?.len() != nbytes as u64 { return Err(std::io::Error::new( std::io::ErrorKind::InvalidData, "TokenSpeed SHM writer produced an unexpected byte length", @@ -424,7 +425,7 @@ pub fn write_tokenspeed_shm_with( } Ok(()) }) { - drop(file); + drop(writer); let _ = remove_file(&path); return Err(error); } From 1959b6092c3a1ae77e6c17b93f8b6c1052e4ab8e Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Tue, 16 Jun 2026 23:42:47 -0700 Subject: [PATCH 08/28] feat(multimodal): auto TokenSpeed MM tensor transport + clearer env names Rename the TokenSpeed multimodal tensor transport env vars to make their multimodal-only scope explicit (prompt input_ids are always sent inline): SMG_TOKENSPEED_TENSOR_TRANSPORT -> SMG_TOKENSPEED_MM_TENSOR_TRANSPORT SMG_TOKENSPEED_SHM_MIN_BYTES -> SMG_TOKENSPEED_MM_SHM_MIN_BYTES The legacy names are still honored as a fallback. Add an `auto` transport mode. The shm vs inline decision is resolved once per request and threaded through to both the encoder input and the model-specific tensors so into_proto no longer re-reads the environment: - shm : always use SHM for tensors >= the size threshold - auto : use SHM only when the worker is local (loopback/unix url and therefore shares SMG's /dev/shm), else inline - inline : never (default) This keeps the safe inline default for distributed/non-colocated workers (the SHM consumer has no fallback when the /dev/shm file is absent) while letting co-located single-host deployments get the SHM path automatically. Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- model_gateway/src/routers/grpc/multimodal.rs | 68 +++++++++++++-- .../src/routers/grpc/proto_wrapper.rs | 84 ++++++++++++------- 2 files changed, 115 insertions(+), 37 deletions(-) diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index ef5e46d185..bf088a2e01 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -39,7 +39,7 @@ use crate::routers::grpc::{ client::GrpcClient, context::WorkerSelection, proto_wrapper::{ - tokenspeed_shm_transport_enabled_for_bytes, write_tokenspeed_shm_with, + tokenspeed_mm_shm_min_bytes, tokenspeed_mm_tensor_transport_mode, write_tokenspeed_shm_with, SglangMultimodalData, TensorBytes, TokenSpeedModality, TokenSpeedMultimodalData, TokenSpeedMultimodalItem, TokenSpeedTensor, TrtllmMultimodalData, VllmMultimodalData, }, @@ -1060,6 +1060,9 @@ fn assemble_tokenspeed( ) -> Result { let log_timing = log_mm_timing_enabled(); let total_started = Instant::now(); + // Resolve the multimodal tensor transport once per request: `shm` always on, + // `auto` only when the worker is local (shares /dev/shm), otherwise inline. + let shm_enabled = resolve_tokenspeed_shm_enabled(workers); // Use patch-only offsets when available and non-empty; fall back to full structural ranges. let encoder_input_dtype = tokenspeed_encoder_input_dtype(intermediate.modality, workers); let patch_offsets = intermediate @@ -1084,8 +1087,11 @@ fn assemble_tokenspeed( item_index, )?; let encoder_input_started = Instant::now(); - let encoder_input = - serialize_array_as_tokenspeed_tensor(&item_encoder_input, &encoder_input_dtype); + let encoder_input = serialize_array_as_tokenspeed_tensor( + &item_encoder_input, + &encoder_input_dtype, + shm_enabled, + ); let encoder_input_serialize_ms = encoder_input_started.elapsed().as_secs_f64() * 1000.0; let model_specific_started = Instant::now(); let model_specific_tensors = serialize_model_specific_for_item( @@ -1134,7 +1140,7 @@ fn assemble_tokenspeed( ); } - Ok(TokenSpeedMultimodalData { items }) + Ok(TokenSpeedMultimodalData { items, shm_enabled }) } fn precomputed_multimodal_item_count( @@ -1348,6 +1354,7 @@ fn serialize_array(encoder_input: &ArrayD) -> (Vec, Vec) { fn serialize_array_as_tokenspeed_tensor( encoder_input: &ArrayD, dtype: &str, + shm_enabled: bool, ) -> TokenSpeedTensor { let dtype = match canonical_float_dtype(dtype).as_deref() { Some("float32") => "float32".to_string(), @@ -1369,7 +1376,7 @@ fn serialize_array_as_tokenspeed_tensor( }; let nbytes = encoder_input.len() * element_size; - if tokenspeed_shm_transport_enabled_for_bytes(nbytes) { + if shm_enabled && nbytes >= tokenspeed_mm_shm_min_bytes() { let started = Instant::now(); match write_tokenspeed_shm_with(nbytes, |file| { write_array_as_dtype(file, encoder_input, &dtype) @@ -1609,6 +1616,57 @@ fn tokenspeed_encoder_input_dtype_from_worker(workers: Option<&WorkerSelection>) .cloned() } +/// Resolve whether large multimodal tensors should use the SHM transport for +/// this request. `shm` = always (legacy explicit opt-in); `auto` = only when the +/// worker is local and therefore shares SMG's `/dev/shm`; anything else +/// (including unset or `inline`) keeps the inline gRPC path. +fn resolve_tokenspeed_shm_enabled(workers: Option<&WorkerSelection>) -> bool { + match tokenspeed_mm_tensor_transport_mode().as_str() { + "shm" => true, + "auto" => worker_is_local(workers), + _ => false, + } +} + +/// A worker is treated as "local" (assumed to share SMG's `/dev/shm`) when its +/// gRPC URL targets a loopback host or a unix-domain socket. Conservative: an +/// unknown worker is non-local so `auto` falls back to the safe inline path. +fn worker_is_local(workers: Option<&WorkerSelection>) -> bool { + let worker = match workers { + Some(WorkerSelection::Single { worker }) => worker, + Some(WorkerSelection::Dual { prefill, .. }) => prefill, + None => return false, + }; + url_is_loopback(worker.url()) +} + +fn url_is_loopback(url: &str) -> bool { + if url.starts_with("unix:") { + return true; + } + let rest = url.split_once("://").map(|(_, r)| r).unwrap_or(url); + let authority = rest.split('/').next().unwrap_or(rest); + let host = if let Some(stripped) = authority.strip_prefix('[') { + // IPv6 literal, e.g. [::1]:47165 + stripped.split(']').next().unwrap_or(stripped) + } else { + authority + .rsplit_once(':') + .map(|(h, _)| h) + .unwrap_or(authority) + } + .trim(); + host == "localhost" + || host + .parse::() + .map(|ip| ip.is_loopback()) + .unwrap_or(false) + || host + .parse::() + .map(|ip| ip.is_loopback()) + .unwrap_or(false) +} + fn canonical_float_dtype(dtype: &str) -> Option { match dtype.trim().to_ascii_lowercase().as_str() { "float32" | "fp32" | "f32" => Some("float32".to_string()), diff --git a/model_gateway/src/routers/grpc/proto_wrapper.rs b/model_gateway/src/routers/grpc/proto_wrapper.rs index 56be720501..2018eea89c 100644 --- a/model_gateway/src/routers/grpc/proto_wrapper.rs +++ b/model_gateway/src/routers/grpc/proto_wrapper.rs @@ -87,6 +87,10 @@ pub struct TrtllmMultimodalData { #[derive(Debug)] pub struct TokenSpeedMultimodalData { pub items: Vec, + /// Resolved per-request decision: may large multimodal tensors use the SHM + /// transport? Computed upstream from the transport mode and (for `auto`) + /// worker locality, so `into_proto` does not re-read the environment. + pub shm_enabled: bool, } #[derive(Debug)] @@ -252,17 +256,18 @@ impl TrtllmMultimodalData { impl TokenSpeedMultimodalData { /// Convert to TokenSpeed proto MultimodalInputs. pub fn into_proto(self) -> tokenspeed::MultimodalInputs { + let shm_enabled = self.shm_enabled; let items = self .items .into_iter() - .map(TokenSpeedMultimodalItem::into_proto) + .map(|item| item.into_proto(shm_enabled)) .collect(); tokenspeed::MultimodalInputs { items } } } impl TokenSpeedMultimodalItem { - fn into_proto(self) -> tokenspeed::MultimodalItem { + fn into_proto(self, shm_enabled: bool) -> tokenspeed::MultimodalItem { let placeholders = self .mm_placeholders .into_iter() @@ -272,7 +277,7 @@ impl TokenSpeedMultimodalItem { let model_specific_tensors = self .model_specific_tensors .into_iter() - .map(|(k, v)| (k, tensor_bytes_to_tokenspeed(v))) + .map(|(k, v)| (k, tensor_bytes_to_tokenspeed(v, shm_enabled))) .collect::>(); tokenspeed::MultimodalItem { @@ -282,7 +287,7 @@ impl TokenSpeedMultimodalItem { TokenSpeedModality::Video => tokenspeed::Modality::Video as i32, }, content_hash: self.content_hash, - encoder_input: Some(tokenspeed_tensor_to_proto(self.encoder_input)), + encoder_input: Some(tokenspeed_tensor_to_proto(self.encoder_input, shm_enabled)), model_specific_tensors, placeholders, placeholder_token_id: self.placeholder_token_id, @@ -290,14 +295,14 @@ impl TokenSpeedMultimodalItem { } } -fn tokenspeed_tensor_to_proto(value: TokenSpeedTensor) -> tokenspeed::TensorData { +fn tokenspeed_tensor_to_proto(value: TokenSpeedTensor, shm_enabled: bool) -> tokenspeed::TensorData { let TokenSpeedTensor { storage, shape, dtype, } = value; let payload = match storage { - TokenSpeedTensorStorage::Inline(data) => tokenspeed_tensor_payload(data), + TokenSpeedTensorStorage::Inline(data) => tokenspeed_tensor_payload(data, shm_enabled), TokenSpeedTensorStorage::Shm(handle) => tokenspeed::tensor_data::Payload::Shm(handle), }; @@ -308,23 +313,19 @@ fn tokenspeed_tensor_to_proto(value: TokenSpeedTensor) -> tokenspeed::TensorData } } -fn tensor_bytes_to_tokenspeed(value: TensorBytes) -> tokenspeed::TensorData { +fn tensor_bytes_to_tokenspeed(value: TensorBytes, shm_enabled: bool) -> tokenspeed::TensorData { let TensorBytes { data, shape, dtype } = value; tokenspeed::TensorData { shape, dtype, - payload: Some(tokenspeed_tensor_payload(data)), + payload: Some(tokenspeed_tensor_payload(data, shm_enabled)), } } -fn tokenspeed_tensor_payload(data: Vec) -> tokenspeed::tensor_data::Payload { +fn tokenspeed_tensor_payload(data: Vec, shm_enabled: bool) -> tokenspeed::tensor_data::Payload { let log_timing = log_tokenspeed_mm_timing_enabled(); - let mode = std::env::var("SMG_TOKENSPEED_TENSOR_TRANSPORT") - .unwrap_or_default() - .trim() - .to_ascii_lowercase(); - if mode != "shm" { + if !shm_enabled { if log_timing { tracing::info!( nbytes = data.len(), @@ -334,10 +335,7 @@ fn tokenspeed_tensor_payload(data: Vec) -> tokenspeed::tensor_data::Payload return tokenspeed::tensor_data::Payload::Inline(data); } - let min_bytes = std::env::var("SMG_TOKENSPEED_SHM_MIN_BYTES") - .ok() - .and_then(|value| value.parse::().ok()) - .unwrap_or(64 * 1024); + let min_bytes = tokenspeed_mm_shm_min_bytes(); if data.len() < min_bytes { if log_timing { tracing::info!( @@ -378,20 +376,30 @@ fn log_tokenspeed_mm_timing_enabled() -> bool { .unwrap_or(false) } -pub fn tokenspeed_shm_transport_enabled_for_bytes(nbytes: usize) -> bool { - let mode = std::env::var("SMG_TOKENSPEED_TENSOR_TRANSPORT") +/// Multimodal tensor transport mode for the TokenSpeed backend. +/// +/// This only governs multimodal tensor payloads (encoder inputs and +/// model-specific tensors); prompt `input_ids` are always sent inline. The +/// canonical env var is `SMG_TOKENSPEED_MM_TENSOR_TRANSPORT`; the legacy +/// `SMG_TOKENSPEED_TENSOR_TRANSPORT` is still honored for backward +/// compatibility. +pub fn tokenspeed_mm_tensor_transport_mode() -> String { + std::env::var("SMG_TOKENSPEED_MM_TENSOR_TRANSPORT") + .or_else(|_| std::env::var("SMG_TOKENSPEED_TENSOR_TRANSPORT")) .unwrap_or_default() .trim() - .to_ascii_lowercase(); - if mode != "shm" { - return false; - } + .to_ascii_lowercase() +} - let min_bytes = std::env::var("SMG_TOKENSPEED_SHM_MIN_BYTES") +/// Minimum multimodal tensor size (bytes) before the SHM transport is used. +/// Canonical env var `SMG_TOKENSPEED_MM_SHM_MIN_BYTES`, with legacy fallback to +/// `SMG_TOKENSPEED_SHM_MIN_BYTES`. Defaults to 64 KiB. +pub fn tokenspeed_mm_shm_min_bytes() -> usize { + std::env::var("SMG_TOKENSPEED_MM_SHM_MIN_BYTES") + .or_else(|_| std::env::var("SMG_TOKENSPEED_SHM_MIN_BYTES")) .ok() .and_then(|value| value.parse::().ok()) - .unwrap_or(64 * 1024); - nbytes >= min_bytes + .unwrap_or(64 * 1024) } static TOKENSPEED_SHM_COUNTER: AtomicU64 = AtomicU64::new(0); @@ -400,6 +408,12 @@ fn write_tokenspeed_shm(data: &[u8]) -> std::io::Result { write_tokenspeed_shm_with(data.len(), |file| file.write_all(data)) } +// TODO: pack all of a request's tensors (encoder_input + model_specific) into +// ONE /dev/shm segment at running offsets instead of one file per tensor +// (ShmHandle.offset already exists, always 0 here). Needs consumer +// ShmTensorHandle offset support + a per-segment refcount so the segment is +// unlinked exactly once after all its tensors are consumed. Cleanliness / fewer +// files, not a measured speed win (tmpfs makes per-file syscalls negligible). pub fn write_tokenspeed_shm_with( nbytes: usize, write_fn: impl FnOnce(&mut BufWriter) -> std::io::Result<()>, @@ -1586,6 +1600,7 @@ mod tests { mm_placeholders: vec![(4, 2)], content_hash: vec![7; 32], }], + shm_enabled: false, } .into_proto(); @@ -1627,6 +1642,7 @@ mod tests { mm_placeholders: vec![(4, 2)], content_hash: vec![7; 32], }], + shm_enabled: false, } .into_proto(); @@ -1645,11 +1661,14 @@ mod tests { #[test] fn tokenspeed_tensor_data_uses_clean_payload_tags() { - let tensor = tensor_bytes_to_tokenspeed(TensorBytes { - data: vec![0xaa, 0xbb], - shape: vec![2, 3], - dtype: "uint32".to_string(), - }); + let tensor = tensor_bytes_to_tokenspeed( + TensorBytes { + data: vec![0xaa, 0xbb], + shape: vec![2, 3], + dtype: "uint32".to_string(), + }, + false, + ); assert_eq!( tensor.encode_to_vec(), @@ -1681,6 +1700,7 @@ mod tests { mm_placeholders: vec![(4, 2)], content_hash: vec![7; 32], }], + shm_enabled: true, } .into_proto(); From 58bfaa1e8d2335e315b42a0f3f78c1656480ce22 Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Wed, 17 Jun 2026 00:29:46 -0700 Subject: [PATCH 09/28] feat(multimodal): sweep orphaned SHM files + harden auto transport Robustness for the TokenSpeed SHM tensor transport: - Sweep orphaned /dev/shm payload files left by a previous SMG process that crashed between writing a segment and the consumer unlinking it. Run once before the first SHM write; only files named smg-tokenspeed--... whose producer pid is no longer alive (and never our own) are removed (pid recycling is a safe miss). Previously only the send-failure path cleaned up. - Gate the SHM path on a one-time /dev/shm writability probe so `auto`/`shm` fall back to inline (with a single warning) when /dev/shm is unmounted/full instead of failing per request. - Validate SMG_TOKENSPEED_MM_TENSOR_TRANSPORT: unknown values now warn once and fall back to inline, and the resolved mode/threshold/writability is logged once. Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- model_gateway/src/routers/grpc/multimodal.rs | 40 ++++++++-- .../src/routers/grpc/proto_wrapper.rs | 78 ++++++++++++++++++- 2 files changed, 110 insertions(+), 8 deletions(-) diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index bf088a2e01..0a58dd108c 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -39,7 +39,8 @@ use crate::routers::grpc::{ client::GrpcClient, context::WorkerSelection, proto_wrapper::{ - tokenspeed_mm_shm_min_bytes, tokenspeed_mm_tensor_transport_mode, write_tokenspeed_shm_with, + tokenspeed_mm_shm_min_bytes, tokenspeed_mm_tensor_transport_mode, + tokenspeed_shm_dev_writable, write_tokenspeed_shm_with, SglangMultimodalData, TensorBytes, TokenSpeedModality, TokenSpeedMultimodalData, TokenSpeedMultimodalItem, TokenSpeedTensor, TrtllmMultimodalData, VllmMultimodalData, }, @@ -1621,13 +1622,42 @@ fn tokenspeed_encoder_input_dtype_from_worker(workers: Option<&WorkerSelection>) /// worker is local and therefore shares SMG's `/dev/shm`; anything else /// (including unset or `inline`) keeps the inline gRPC path. fn resolve_tokenspeed_shm_enabled(workers: Option<&WorkerSelection>) -> bool { - match tokenspeed_mm_tensor_transport_mode().as_str() { - "shm" => true, - "auto" => worker_is_local(workers), - _ => false, + let mode = tokenspeed_mm_tensor_transport_mode(); + log_tokenspeed_transport_config_once(&mode); + match mode.as_str() { + // SHM only ever happens when SMG can actually write /dev/shm. + "shm" => tokenspeed_shm_dev_writable(), + "auto" => worker_is_local(workers) && tokenspeed_shm_dev_writable(), + "" | "inline" => false, + other => { + log_unknown_tokenspeed_transport_once(other); + false + } } } +fn log_tokenspeed_transport_config_once(mode: &str) { + static LOGGED: OnceLock<()> = OnceLock::new(); + LOGGED.get_or_init(|| { + info!( + mode, + shm_min_bytes = tokenspeed_mm_shm_min_bytes(), + dev_writable = tokenspeed_shm_dev_writable(), + "TokenSpeed multimodal tensor transport configured" + ); + }); +} + +fn log_unknown_tokenspeed_transport_once(value: &str) { + static WARNED: OnceLock<()> = OnceLock::new(); + WARNED.get_or_init(|| { + warn!( + value, + "Unknown SMG_TOKENSPEED_MM_TENSOR_TRANSPORT value; expected inline|shm|auto, using inline" + ); + }); +} + /// A worker is treated as "local" (assumed to share SMG's `/dev/shm`) when its /// gRPC URL targets a loopback host or a unix-domain socket. Conservative: an /// unknown worker is non-local so `auto` falls back to the safe inline path. diff --git a/model_gateway/src/routers/grpc/proto_wrapper.rs b/model_gateway/src/routers/grpc/proto_wrapper.rs index 2018eea89c..3e960e6a3f 100644 --- a/model_gateway/src/routers/grpc/proto_wrapper.rs +++ b/model_gateway/src/routers/grpc/proto_wrapper.rs @@ -6,11 +6,14 @@ use std::{ collections::HashMap, - fs::{remove_file, OpenOptions}, + fs::{read_dir, remove_file, OpenOptions}, io::{BufWriter, Write}, - path::PathBuf, + path::{Path, PathBuf}, process, - sync::atomic::{AtomicU64, Ordering}, + sync::{ + atomic::{AtomicU64, Ordering}, + OnceLock, + }, time::{Instant, SystemTime, UNIX_EPOCH}, }; @@ -408,6 +411,74 @@ fn write_tokenspeed_shm(data: &[u8]) -> std::io::Result { write_tokenspeed_shm_with(data.len(), |file| file.write_all(data)) } +/// Whether SMG can actually create+write files under `/dev/shm`. Probed once; +/// when false the SHM transport cannot work, so `auto`/`shm` must stay inline. +pub fn tokenspeed_shm_dev_writable() -> bool { + static WRITABLE: OnceLock = OnceLock::new(); + *WRITABLE.get_or_init(|| { + let name = format!("smg-tokenspeed-probe-{}", process::id()); + let path = tokenspeed_shm_path(&name); + let ok = OpenOptions::new() + .write(true) + .create(true) + .truncate(true) + .open(&path) + .and_then(|mut file| file.write_all(b"x")) + .is_ok(); + let _ = remove_file(&path); + if !ok { + tracing::warn!( + path = %path.display(), + "/dev/shm is not writable; TokenSpeed SHM tensor transport will fall back to inline" + ); + } + ok + }) +} + +/// Best-effort, run-once sweep of `/dev/shm` for TokenSpeed payload files left +/// behind by a *previous* SMG process that crashed between writing a segment and +/// the consumer unlinking it. Files are named `smg-tokenspeed--...`; we only +/// remove those whose producer pid is no longer alive (and never our own). +fn sweep_orphan_tokenspeed_shm_once() { + static SWEEP: OnceLock<()> = OnceLock::new(); + SWEEP.get_or_init(|| { + let dir = Path::new("/dev/shm"); + let Ok(entries) = read_dir(dir) else { + return; + }; + let my_pid = process::id(); + let mut removed = 0u32; + for entry in entries.flatten() { + let file_name = entry.file_name(); + let Some(name) = file_name.to_str() else { + continue; + }; + let Some(rest) = name.strip_prefix("smg-tokenspeed-") else { + continue; + }; + // pid is the first '-'-separated field after the prefix. + let Some(pid) = rest.split('-').next().and_then(|p| p.parse::().ok()) else { + continue; + }; + // Skip our own files and any still-live producer (pid recycling is a + // safe miss: we just keep the file rather than risk deleting a live one). + if pid == my_pid || Path::new(&format!("/proc/{pid}")).exists() { + continue; + } + if remove_file(dir.join(name)).is_ok() { + removed += 1; + } + } + if removed > 0 { + tracing::warn!( + count = removed, + "Swept orphaned TokenSpeed SHM files from dead producer processes" + ); + } + }); +} + // TODO: pack all of a request's tensors (encoder_input + model_specific) into // ONE /dev/shm segment at running offsets instead of one file per tensor // (ShmHandle.offset already exists, always 0 here). Needs consumer @@ -418,6 +489,7 @@ pub fn write_tokenspeed_shm_with( nbytes: usize, write_fn: impl FnOnce(&mut BufWriter) -> std::io::Result<()>, ) -> std::io::Result { + sweep_orphan_tokenspeed_shm_once(); let name = next_tokenspeed_shm_name(); let path = tokenspeed_shm_path(&name); let file = OpenOptions::new() From 211206600cf871c800bb718035d5832bfb5cbf04 Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Wed, 17 Jun 2026 01:20:24 -0700 Subject: [PATCH 10/28] feat(multimodal): Prometheus metrics for TokenSpeed MM tensor transport Add counters so the shm-vs-inline transport split is observable: - smg_tokenspeed_mm_tensors_total{path} tensors sent per path - smg_tokenspeed_mm_tensor_bytes_total{path} bytes sent per path - smg_tokenspeed_mm_shm_write_failures_total SHM writes that fell back to inline Each tensor is metered exactly once: encoder inputs written directly to SHM are counted at proto conversion (Shm storage arm); all other tensors (inline-storage encoders + model_specific) are counted inside tokenspeed_tensor_payload. Exposed on the existing Prometheus endpoint (:29000/metrics). Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- model_gateway/src/observability/metrics.rs | 26 ++++++++++++++++ model_gateway/src/routers/grpc/multimodal.rs | 1 + .../src/routers/grpc/proto_wrapper.rs | 30 +++++++++++++------ 3 files changed, 48 insertions(+), 9 deletions(-) diff --git a/model_gateway/src/observability/metrics.rs b/model_gateway/src/observability/metrics.rs index 2701c95efd..457ae3d9a1 100644 --- a/model_gateway/src/observability/metrics.rs +++ b/model_gateway/src/observability/metrics.rs @@ -389,6 +389,20 @@ pub(crate) fn init_metrics() { ); describe_counter!("smg_db_items_stored", "Total items stored by storage_type"); + // TokenSpeed multimodal tensor transport (shm vs inline) + describe_counter!( + "smg_tokenspeed_mm_tensors_total", + "TokenSpeed multimodal tensors sent, by transport path (shm/inline)" + ); + describe_counter!( + "smg_tokenspeed_mm_tensor_bytes_total", + "TokenSpeed multimodal tensor bytes sent, by transport path (shm/inline)" + ); + describe_counter!( + "smg_tokenspeed_mm_shm_write_failures_total", + "TokenSpeed SHM tensor write attempts that failed and fell back to inline" + ); + // Layer 0: Tokio runtime self-observability (event-loop canary + sampler). super::runtime_metrics::describe(); @@ -632,6 +646,18 @@ impl Metrics { .increment(1); } + /// Record one TokenSpeed multimodal tensor sent over `path` ("shm"|"inline"). + pub fn record_tokenspeed_mm_tensor(path: &'static str, nbytes: usize) { + counter!("smg_tokenspeed_mm_tensors_total", "path" => path).increment(1); + counter!("smg_tokenspeed_mm_tensor_bytes_total", "path" => path) + .increment(nbytes as u64); + } + + /// Record a TokenSpeed SHM write that failed and fell back to inline. + pub fn record_tokenspeed_mm_shm_write_failure() { + counter!("smg_tokenspeed_mm_shm_write_failures_total").increment(1); + } + // ======================================================================== // Layer 2: Router metrics // ======================================================================== diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index 0a58dd108c..f5852d1be8 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -1399,6 +1399,7 @@ fn serialize_array_as_tokenspeed_tensor( dtype = %dtype, "Failed to write TokenSpeed encoder input directly to SHM; falling back to bytes path" ); + crate::observability::metrics::Metrics::record_tokenspeed_mm_shm_write_failure(); } } } diff --git a/model_gateway/src/routers/grpc/proto_wrapper.rs b/model_gateway/src/routers/grpc/proto_wrapper.rs index 3e960e6a3f..4e93d8505d 100644 --- a/model_gateway/src/routers/grpc/proto_wrapper.rs +++ b/model_gateway/src/routers/grpc/proto_wrapper.rs @@ -305,8 +305,16 @@ fn tokenspeed_tensor_to_proto(value: TokenSpeedTensor, shm_enabled: bool) -> tok dtype, } = value; let payload = match storage { + // Inline storage is metered inside tokenspeed_tensor_payload. TokenSpeedTensorStorage::Inline(data) => tokenspeed_tensor_payload(data, shm_enabled), - TokenSpeedTensorStorage::Shm(handle) => tokenspeed::tensor_data::Payload::Shm(handle), + // Encoder input already written directly to SHM upstream — meter it here. + TokenSpeedTensorStorage::Shm(handle) => { + crate::observability::metrics::Metrics::record_tokenspeed_mm_tensor( + "shm", + handle.nbytes as usize, + ); + tokenspeed::tensor_data::Payload::Shm(handle) + } }; tokenspeed::TensorData { @@ -327,26 +335,27 @@ fn tensor_bytes_to_tokenspeed(value: TensorBytes, shm_enabled: bool) -> tokenspe } fn tokenspeed_tensor_payload(data: Vec, shm_enabled: bool) -> tokenspeed::tensor_data::Payload { + use crate::observability::metrics::Metrics; let log_timing = log_tokenspeed_mm_timing_enabled(); + let nbytes = data.len(); if !shm_enabled { if log_timing { - tracing::info!( - nbytes = data.len(), - "smg_mm_timing tokenspeed_tensor_payload_inline" - ); + tracing::info!(nbytes, "smg_mm_timing tokenspeed_tensor_payload_inline"); } + Metrics::record_tokenspeed_mm_tensor("inline", nbytes); return tokenspeed::tensor_data::Payload::Inline(data); } let min_bytes = tokenspeed_mm_shm_min_bytes(); - if data.len() < min_bytes { + if nbytes < min_bytes { if log_timing { tracing::info!( - nbytes = data.len(), + nbytes, min_bytes, "smg_mm_timing tokenspeed_tensor_payload_inline_below_threshold" ); } + Metrics::record_tokenspeed_mm_tensor("inline", nbytes); return tokenspeed::tensor_data::Payload::Inline(data); } @@ -355,19 +364,22 @@ fn tokenspeed_tensor_payload(data: Vec, shm_enabled: bool) -> tokenspeed::te Ok(handle) => { if log_timing { tracing::info!( - nbytes = data.len(), + nbytes, elapsed_ms = started.elapsed().as_secs_f64() * 1000.0, "smg_mm_timing tokenspeed_shm_write" ); } + Metrics::record_tokenspeed_mm_tensor("shm", nbytes); tokenspeed::tensor_data::Payload::Shm(handle) } Err(error) => { tracing::warn!( ?error, - nbytes = data.len(), + nbytes, "Failed to create TokenSpeed SHM tensor payload; falling back to inline" ); + Metrics::record_tokenspeed_mm_shm_write_failure(); + Metrics::record_tokenspeed_mm_tensor("inline", nbytes); tokenspeed::tensor_data::Payload::Inline(data) } } From 9ad1be0759d982ba48bcacff405f8f744aaba5c0 Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Wed, 17 Jun 2026 05:32:03 -0700 Subject: [PATCH 11/28] fix(multimodal): default Qwen VL image resize to bicubic Qwen2VL/Qwen3VL HF image processors default to BICUBIC (PIL resample=3) when the preprocessor config omits `resample`, but SMG's pil_to_filter fell back to bilinear (Triangle). Bilinear produces smoother encoder-input features; against vLLM on Qwen3.5-397B-A17B-NVFP4 / MMBench the SMG pixel_values matched HF only at corr 0.9994 (max abs diff 0.23). Pinning bicubic for the Qwen path raises the match to corr 0.99996 (max 0.04), aligning SMG with the reference HF/vLLM preprocessing. Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- .../src/vision/processors/qwen_vl_base.rs | 18 +++++++++++++++--- 1 file changed, 15 insertions(+), 3 deletions(-) diff --git a/crates/multimodal/src/vision/processors/qwen_vl_base.rs b/crates/multimodal/src/vision/processors/qwen_vl_base.rs index 85831bdf2a..89416aa762 100644 --- a/crates/multimodal/src/vision/processors/qwen_vl_base.rs +++ b/crates/multimodal/src/vision/processors/qwen_vl_base.rs @@ -541,7 +541,11 @@ impl VisionPreProcessor for QwenVLProcessorBase { let mean = config.get_image_mean(); let std = config.get_image_std(); - let filter = pil_to_filter(config.resampling); + // Qwen2VL/Qwen3VL image processors default to BICUBIC (PIL resample=3) + // when the preprocessor config omits `resample`. The global pil_to_filter + // fallback is bilinear, which yields smoother features and measurably + // degrades VLM accuracy, so pin the HF-correct default here. + let filter = pil_to_filter(config.resampling.or(Some(3))); let patch_size = self.config.patch_size; let temporal_patch_size = self.config.temporal_patch_size; @@ -637,7 +641,11 @@ impl VisionPreProcessor for QwenVLProcessorBase { let item_sizes = vec![(w, h)]; let mean = config.get_image_mean(); let std = config.get_image_std(); - let filter = pil_to_filter(config.resampling); + // Qwen2VL/Qwen3VL image processors default to BICUBIC (PIL resample=3) + // when the preprocessor config omits `resample`. The global pil_to_filter + // fallback is bilinear, which yields smoother features and measurably + // degrades VLM accuracy, so pin the HF-correct default here. + let filter = pil_to_filter(config.resampling.or(Some(3))); let temporal_patch_size = self.config.temporal_patch_size; let padded_frames = frames.len().div_ceil(temporal_patch_size) * temporal_patch_size; @@ -745,7 +753,11 @@ impl VisionPreProcessor for QwenVLProcessorBase { let item_sizes = vec![(w, h)]; let mean = config.get_image_mean(); let std = config.get_image_std(); - let filter = pil_to_filter(config.resampling); + // Qwen2VL/Qwen3VL image processors default to BICUBIC (PIL resample=3) + // when the preprocessor config omits `resample`. The global pil_to_filter + // fallback is bilinear, which yields smoother features and measurably + // degrades VLM accuracy, so pin the HF-correct default here. + let filter = pil_to_filter(config.resampling.or(Some(3))); let temporal_patch_size = self.config.temporal_patch_size; let padded_frames = frames.len().div_ceil(temporal_patch_size) * temporal_patch_size; From dc9dda14715e2fa7ef720dfa5262bcda2a5942d2 Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 06:48:20 -0700 Subject: [PATCH 12/28] fix(mm): place image/video placeholders before text in chat content MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The eval harness sends multimodal content as [text, image]. SMG rendered the content parts in order, so the image landed AFTER the whole question — which measurably degrades VQA grounding (MMBench answers flip vs. image-first). vLLM always prepends media placeholders to the front (its default interleave_mm_strings=false); match that. transform_content_field now emits media (image/video/audio) before text in both content formats: OpenAI uses a stable partition; String collects placeholders first then text, joined by "\n". No-op for text-only/string content and content already media-first. A TODO(interleave) documents how to thread vLLM's interleave_mm_strings opt-out if ever needed. Verified: 10 unit tests plus an e2e render test against the real Qwen3.5 chat_template.jinja (image now at char 17 < question at char 60). End-to-end MMBench_DEV_EN_V11 (150-q sample) recovered 87% -> 96%. Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- .../src/routers/grpc/utils/chat_utils.rs | 251 ++++++++++++++++-- 1 file changed, 226 insertions(+), 25 deletions(-) diff --git a/model_gateway/src/routers/grpc/utils/chat_utils.rs b/model_gateway/src/routers/grpc/utils/chat_utils.rs index a83324105a..cc92a35531 100644 --- a/model_gateway/src/routers/grpc/utils/chat_utils.rs +++ b/model_gateway/src/routers/grpc/utils/chat_utils.rs @@ -203,6 +203,20 @@ pub(crate) fn process_content_format( /// being stripped. This mirrors vLLM's behavior of injecting model-specific /// placeholder tokens (e.g. `"<|image|>"`) so that the tokenizer produces /// token IDs the multimodal expansion step can find and replace. +/// +/// Media parts are always emitted BEFORE text, matching vLLM's default +/// (`interleave_mm_strings=false`), which prepends media placeholders to the +/// front of the prompt for every model. This is required for VQA accuracy: +/// the harness sends `[text, image]` and image-after-question measurably +/// degrades grounding (e.g. MMBench answers flip vs. image-first). +/// +/// TODO(interleave): vLLM also supports `interleave_mm_strings=true` (opt-in, +/// only with `--chat-template-content-format=string`), where placeholders stay +/// in their authored position so users can fully interleave media within text. +/// We currently always hoist media to the front and do not expose that opt-out. +/// If/when we need interleaved prompts, thread an `interleave` flag through +/// `process_content_format` (and the router config) and skip the reordering +/// below when it is set, leaving the parts in their original order. fn transform_content_field( content_value: &mut Value, content_format: ChatTemplateContentFormat, @@ -214,23 +228,34 @@ fn transform_content_field( match content_format { ChatTemplateContentFormat::String => { - // Extract text parts; optionally replace image parts with placeholders - let text_parts: Vec = content_array - .iter() - .filter_map(|part| { - let obj = part.as_object()?; - let type_str = obj.get("type")?.as_str()?; - match type_str { - "text" => obj.get("text")?.as_str().map(String::from), - "image_url" => image_placeholder.map(String::from), - "video_url" => image_placeholder.map(String::from), - _ => None, + // Extract text parts; replace media parts with placeholders. Media + // placeholders are emitted FIRST (before text), matching vLLM's + // `_get_full_multimodal_text_prompt` ("always add missing + // placeholders at the front"). Placing the image after the question + // text measurably degrades VQA accuracy (MMBench answers flip). + let mut media_parts: Vec = Vec::new(); + let mut text_parts: Vec = Vec::new(); + for part in content_array { + let Some(obj) = part.as_object() else { continue }; + match obj.get("type").and_then(|t| t.as_str()) { + Some("text") => { + if let Some(t) = obj.get("text").and_then(|t| t.as_str()) { + text_parts.push(t.to_string()); + } } - }) - .collect(); + Some("image_url") | Some("video_url") => { + if let Some(ph) = image_placeholder { + media_parts.push(ph.to_string()); + } + } + _ => {} + } + } - if !text_parts.is_empty() { - *content_value = Value::String(text_parts.join("\n")); + if !media_parts.is_empty() || !text_parts.is_empty() { + let ordered: Vec = + media_parts.into_iter().chain(text_parts).collect(); + *content_value = Value::String(ordered.join("\n")); } } ChatTemplateContentFormat::OpenAI => { @@ -250,7 +275,22 @@ fn transform_content_field( }) .collect(); - *content_value = Value::Array(processed_parts); + // Place media parts before the remaining (text) parts, matching + // vLLM's front placement. The chat template renders parts in order, + // so an OpenAI request like [text, image] would otherwise put the + // image AFTER the whole question — which measurably hurts visual + // grounding (MMBench answers flip vs. image-first). `partition` is + // stable, so relative order within media and within text is kept. + let (mut media, rest): (Vec, Vec) = + processed_parts.into_iter().partition(|p| { + matches!( + p.get("type").and_then(|t| t.as_str()), + Some("image") | Some("video") | Some("audio") + ) + }); + media.extend(rest); + + *content_value = Value::Array(media); } } } @@ -695,9 +735,10 @@ mod tests { ) .unwrap(); + // Media placeholder is emitted before the text (vLLM front placement). assert_eq!( result[0]["content"].as_str().unwrap(), - "Watch this\n<|video|>" + "<|video|>\nWatch this" ); } @@ -724,16 +765,18 @@ mod tests { assert_eq!(result.len(), 1); let transformed_message = &result[0]; - // Should replace media URLs with simple type placeholders + // Media URLs replaced with simple type placeholders, and the image is + // hoisted before the text (vLLM front placement; image-after-question + // degrades VQA accuracy). let content_array = transformed_message["content"].as_array().unwrap(); assert_eq!(content_array.len(), 2); - // Text part should remain unchanged - assert_eq!(content_array[0]["type"], "text"); - assert_eq!(content_array[0]["text"], "Describe this image:"); + // Image part comes first now. + assert_eq!(content_array[0], json!({"type": "image"})); - // Image part should be replaced with simple type placeholder - assert_eq!(content_array[1], json!({"type": "image"})); + // Text part follows, unchanged. + assert_eq!(content_array[1]["type"], "text"); + assert_eq!(content_array[1]["text"], "Describe this image:"); } #[test] @@ -853,7 +896,165 @@ mod tests { let content_array = result_openai[1]["content"].as_array().unwrap(); assert_eq!(content_array.len(), 2); - assert_eq!(content_array[0]["type"], "text"); - assert_eq!(content_array[1], json!({"type": "image"})); + // Image hoisted before text. + assert_eq!(content_array[0], json!({"type": "image"})); + assert_eq!(content_array[1]["type"], "text"); + } + + #[test] + fn test_media_hoisted_before_text_openai() { + // Real MMBench shape: [question text, image] must render image-first. + let messages = vec![ChatMessage::User { + content: MessageContent::Parts(vec![ + ContentPart::Text { + text: "Question: ...\nAnswer with only the option letter.".to_string(), + }, + ContentPart::ImageUrl { + image_url: ImageUrl { + url: "data:image/jpeg;base64,XXX".to_string(), + detail: None, + }, + }, + ]), + name: None, + }]; + + let result = + process_content_format(&messages, ChatTemplateContentFormat::OpenAI, None).unwrap(); + let arr = result[0]["content"].as_array().unwrap(); + assert_eq!(arr[0], json!({"type": "image"})); + assert_eq!(arr[1]["type"], "text"); + } + + #[test] + fn test_media_hoisted_before_text_string() { + // String-format template: placeholder prepended, matching vLLM exactly. + let messages = vec![ChatMessage::User { + content: MessageContent::Parts(vec![ + ContentPart::Text { + text: "Question?".to_string(), + }, + ContentPart::ImageUrl { + image_url: ImageUrl { + url: "data:image/jpeg;base64,XXX".to_string(), + detail: None, + }, + }, + ]), + name: None, + }]; + + let result = process_content_format( + &messages, + ChatTemplateContentFormat::String, + Some("<|image_pad|>"), + ) + .unwrap(); + assert_eq!( + result[0]["content"].as_str().unwrap(), + "<|image_pad|>\nQuestion?" + ); + } + + #[test] + fn test_media_first_stable_and_multi() { + // Multiple media + text keep relative order within each group, media first. + let messages = vec![ChatMessage::User { + content: MessageContent::Parts(vec![ + ContentPart::Text { + text: "a".to_string(), + }, + ContentPart::ImageUrl { + image_url: ImageUrl { + url: "i1".to_string(), + detail: None, + }, + }, + ContentPart::Text { + text: "b".to_string(), + }, + ContentPart::ImageUrl { + image_url: ImageUrl { + url: "i2".to_string(), + detail: None, + }, + }, + ]), + name: None, + }]; + + let result = + process_content_format(&messages, ChatTemplateContentFormat::OpenAI, None).unwrap(); + let arr = result[0]["content"].as_array().unwrap(); + assert_eq!(arr[0], json!({"type": "image"})); + assert_eq!(arr[1], json!({"type": "image"})); + assert_eq!(arr[2]["text"], "a"); + assert_eq!(arr[3]["text"], "b"); + } + + /// End-to-end: run a real MMBench-shaped `[text, image]` message through the + /// full SMG pipeline (`process_content_format` + the actual model chat + /// template) and assert the rendered prompt places the image BEFORE the + /// question. Proves the fix end-to-end without needing the GPU worker. + /// Ignored by default (needs the real template); run with: + /// QWEN35_CHAT_TEMPLATE=/path/chat_template.jinja cargo test -p smg \ + /// render_image_before_question -- --ignored --nocapture + #[test] + #[ignore = "needs real chat_template.jinja via QWEN35_CHAT_TEMPLATE"] + fn test_render_image_before_question_real_template() { + use llm_tokenizer::chat_template::{ + detect_chat_template_content_format, ChatTemplateParams, ChatTemplateProcessor, + }; + + let Ok(path) = std::env::var("QWEN35_CHAT_TEMPLATE") else { + return; // skip when not provided + }; + let template = std::fs::read_to_string(&path).expect("read template"); + let format = detect_chat_template_content_format(&template); + eprintln!("detected content format: {format}"); + + let messages = vec![ChatMessage::User { + content: MessageContent::Parts(vec![ + ContentPart::Text { + text: "Question: Which description is correct?\n\ + Answer with only the option letter (A/B/C/D)." + .to_string(), + }, + ContentPart::ImageUrl { + image_url: ImageUrl { + url: "data:image/jpeg;base64,XXX".to_string(), + detail: None, + }, + }, + ]), + name: None, + }]; + + let transformed = + process_content_format(&messages, format, Some("<|image_pad|>")).unwrap(); + + let mut kwargs = std::collections::HashMap::new(); + kwargs.insert("enable_thinking".to_string(), json!(false)); + let params = ChatTemplateParams { + add_generation_prompt: true, + template_kwargs: Some(&kwargs), + ..Default::default() + }; + let rendered = ChatTemplateProcessor::new(template) + .unwrap() + .apply_chat_template(&transformed, params) + .unwrap(); + + let vstart = rendered + .find("<|vision_start|>") + .expect("rendered prompt has <|vision_start|>"); + let qpos = rendered + .find("Question:") + .expect("rendered prompt has the question"); + eprintln!("vision_start@{vstart} question@{qpos}"); + assert!( + vstart < qpos, + "image must precede the question (vstart={vstart}, qpos={qpos}).\n--- rendered ---\n{rendered}" + ); } } From 646e93441da701229e6fe7c082e3dd060fb2cb23 Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 07:26:06 -0700 Subject: [PATCH 13/28] perf(mm): parallelize Pillow-exact BICUBIC resize (bit-identical) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Pillow-exact BICUBIC resize (resize_bicubic_pil), used for vLLM/PIL preprocessing parity, ran both passes single-threaded. On large images its scalar fixed-point arithmetic dominated preprocessing (≈4x slower than HF on 12MP). Each output row is an independent fixed-point integer sum, so band the horizontal/vertical passes across threads with std::thread::scope: identical arithmetic and inner-sum order => BIT-IDENTICAL output, no new dependency. Small images stay serial (par_threads gates on output size / rows-per-thread) to avoid spawn overhead. Accuracy is preserved unconditionally: resize_fingerprint.rs pins the exact byte output (fnv1a) of the serial implementation across up/downscale cases and asserts the parallel version reproduces it bit-for-bit. Golden parity (qwen35_parity) diffs vs HF are unchanged. Measured (Qwen3-VL preprocess, 224-core host): 12MP 862 -> 345 ms (2.5x), small/medium unchanged (resize not the hotspot there). Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- crates/multimodal/src/vision/transforms.rs | 278 ++++++++++++++++++ crates/multimodal/tests/resize_fingerprint.rs | 68 +++++ 2 files changed, 346 insertions(+) create mode 100644 crates/multimodal/tests/resize_fingerprint.rs diff --git a/crates/multimodal/src/vision/transforms.rs b/crates/multimodal/src/vision/transforms.rs index deb68fa218..415fdd5d34 100644 --- a/crates/multimodal/src/vision/transforms.rs +++ b/crates/multimodal/src/vision/transforms.rs @@ -286,6 +286,284 @@ fn fir_image_to_dynamic( .unwrap_or_else(|| source.resize_exact(width, height, filter)) } +// --------------------------------------------------------------------------- +// Pillow-exact bicubic resize. +// +// HuggingFace's (slow) `Qwen2VLImageProcessor` — which vLLM uses for Qwen2/3-VL +// — resizes via `PIL.Image.resize(size, BICUBIC)` on the uint8 image. The SIMD +// `fast_image_resize` path above is the same filter *family* (Catmull-Rom, +// a=-0.5) but diverges bit-wise on non-integer ratios (support scaling + +// fixed-point details), which the vision encoder amplifies into a large +// embedding shift vs vLLM. This routine replicates Pillow's `Resample.c` +// algorithm exactly (validated bit-for-bit against Pillow) so SMG's encoder +// inputs match HF/vLLM. +const PIL_PRECISION_BITS: i64 = 32 - 8 - 2; +const PIL_BICUBIC_SUPPORT: f64 = 2.0; + +#[inline] +fn pil_cubic(x: f64) -> f64 { + // Keys cubic with a = -0.5 (Pillow's BICUBIC). + const A: f64 = -0.5; + let x = x.abs(); + if x < 1.0 { + ((A + 2.0) * x - (A + 3.0)) * x * x + 1.0 + } else if x < 2.0 { + (((x - 5.0) * x + 8.0) * x - 4.0) * A + } else { + 0.0 + } +} + +/// Pillow `precompute_coeffs` for one axis: integer (fixed-point) kernels plus +/// per-output bounds `(start, count)`. +fn pil_precompute_coeffs(in_size: usize, out_size: usize) -> (Vec<(usize, usize)>, Vec>) { + let scale = in_size as f64 / out_size as f64; + let filterscale = if scale >= 1.0 { scale } else { 1.0 }; + let support = PIL_BICUBIC_SUPPORT * filterscale; + let inv = 1.0 / filterscale; + let coeff_scale = (1_i64 << PIL_PRECISION_BITS) as f64; + + let mut bounds = Vec::with_capacity(out_size); + let mut kernels = Vec::with_capacity(out_size); + for xx in 0..out_size { + let center = (xx as f64 + 0.5) * scale; + let mut xmin = (center - support + 0.5) as i64; + if xmin < 0 { + xmin = 0; + } + let mut xmax = (center + support + 0.5) as i64; + if xmax > in_size as i64 { + xmax = in_size as i64; + } + let xmin = xmin as usize; + let xmax = (xmax as usize).saturating_sub(xmin); + + let mut w = vec![0.0_f64; xmax]; + let mut tot = 0.0; + for (x, wx) in w.iter_mut().enumerate() { + let v = pil_cubic(((x + xmin) as f64 - center + 0.5) * inv); + *wx = v; + tot += v; + } + if tot != 0.0 { + for wx in &mut w { + *wx /= tot; + } + } + // Pillow normalize_coeffs_8bpc: round half away from zero into fixed point. + let k: Vec = w + .iter() + .map(|&c| { + if c < 0.0 { + (-0.5 + c * coeff_scale) as i64 + } else { + (0.5 + c * coeff_scale) as i64 + } + }) + .collect(); + bounds.push((xmin, xmax)); + kernels.push(k); + } + (bounds, kernels) +} + +#[inline] +fn pil_clip8(v: i64) -> u8 { + let v = v >> PIL_PRECISION_BITS; + if v < 0 { + 0 + } else if v > 255 { + 255 + } else { + v as u8 + } +} + +/// Number of threads to split a resample pass across. Each output row is an +/// independent fixed-point integer sum, so banding rows over threads yields +/// BIT-IDENTICAL output (no shared accumulation, inner sum order unchanged). +/// Small images run serial to avoid thread-spawn overhead. +fn par_threads(out_bytes: usize, out_rows: usize) -> usize { + const PAR_MIN_BYTES: usize = 1 << 19; // ~512 KiB output; below this, serial + const MIN_ROWS_PER_THREAD: usize = 32; // keep enough work per thread + const MAX_THREADS: usize = 32; // spawning hundreds of threads costs more than it saves + if out_bytes < PAR_MIN_BYTES || out_rows < 2 * MIN_ROWS_PER_THREAD { + return 1; + } + let avail = std::thread::available_parallelism() + .map(|n| n.get()) + .unwrap_or(1); + (out_rows / MIN_ROWS_PER_THREAD) + .min(avail) + .min(MAX_THREADS) + .max(1) +} + +/// Process output rows `[oy0, oy0 + out_band.len()/row_out)` of the horizontal +/// pass into `out_band`. Horizontal pass preserves row count, so output row i +/// reads input row `oy0 + i`. +#[allow(clippy::too_many_arguments)] +fn pil_h_band( + src: &[u8], + bounds: &[(usize, usize)], + kernels: &[Vec], + half: i64, + in_w: usize, + out_w: usize, + channels: usize, + oy0: usize, + out_band: &mut [u8], +) { + let row_out = out_w * channels; + for (i, orow) in out_band.chunks_mut(row_out).enumerate() { + let y = oy0 + i; + let row = &src[y * in_w * channels..(y + 1) * in_w * channels]; + for xx in 0..out_w { + let (xmin, xmax) = bounds[xx]; + let k = &kernels[xx]; + for c in 0..channels { + let mut ss = half; + for x in 0..xmax { + ss += row[(xmin + x) * channels + c] as i64 * k[x]; + } + orow[xx * channels + c] = pil_clip8(ss); + } + } + } +} + +/// Resample interleaved `channels`-channel u8 data along the width axis. +/// `src` is `rows * in_w * channels`; returns `rows * out_w * channels`. +fn pil_resample_horizontal( + src: &[u8], + rows: usize, + in_w: usize, + out_w: usize, + channels: usize, +) -> Vec { + let (bounds, kernels) = pil_precompute_coeffs(in_w, out_w); + let half = 1_i64 << (PIL_PRECISION_BITS - 1); + let row_out = out_w * channels; + let mut out = vec![0_u8; rows * row_out]; + let nthreads = par_threads(out.len(), rows); + if nthreads <= 1 { + pil_h_band(src, &bounds, &kernels, half, in_w, out_w, channels, 0, &mut out); + } else { + let chunk_rows = rows.div_ceil(nthreads); + std::thread::scope(|s| { + let (b, k) = (&bounds, &kernels); + let mut rest = out.as_mut_slice(); + let mut oy0 = 0usize; + while oy0 < rows { + let n = chunk_rows.min(rows - oy0); + let (band, tail) = rest.split_at_mut(n * row_out); + rest = tail; + let start = oy0; + s.spawn(move || { + pil_h_band(src, b, k, half, in_w, out_w, channels, start, band) + }); + oy0 += n; + } + }); + } + out +} + +/// Process output rows `[oy0, oy0 + out_band.len()/row_out)` of the vertical +/// pass into `out_band`. +#[allow(clippy::too_many_arguments)] +fn pil_v_band( + src: &[u8], + bounds: &[(usize, usize)], + kernels: &[Vec], + half: i64, + width: usize, + channels: usize, + oy0: usize, + out_band: &mut [u8], +) { + let row_out = width * channels; + for (i, orow) in out_band.chunks_mut(row_out).enumerate() { + let yy = oy0 + i; + let (ymin, ymax) = bounds[yy]; + let k = &kernels[yy]; + for x in 0..width { + for c in 0..channels { + let mut ss = half; + for y in 0..ymax { + ss += src[((ymin + y) * width + x) * channels + c] as i64 * k[y]; + } + orow[x * channels + c] = pil_clip8(ss); + } + } + } +} + +/// Resample interleaved `channels`-channel u8 data along the height axis. +fn pil_resample_vertical( + src: &[u8], + in_h: usize, + width: usize, + out_h: usize, + channels: usize, +) -> Vec { + let (bounds, kernels) = pil_precompute_coeffs(in_h, out_h); + let half = 1_i64 << (PIL_PRECISION_BITS - 1); + let row_out = width * channels; + let mut out = vec![0_u8; out_h * row_out]; + let nthreads = par_threads(out.len(), out_h); + if nthreads <= 1 { + pil_v_band(src, &bounds, &kernels, half, width, channels, 0, &mut out); + } else { + let chunk_rows = out_h.div_ceil(nthreads); + std::thread::scope(|s| { + let (b, k) = (&bounds, &kernels); + let mut rest = out.as_mut_slice(); + let mut oy0 = 0usize; + while oy0 < out_h { + let n = chunk_rows.min(out_h - oy0); + let (band, tail) = rest.split_at_mut(n * row_out); + rest = tail; + let start = oy0; + s.spawn(move || pil_v_band(src, b, k, half, width, channels, start, band)); + oy0 += n; + } + }); + } + out +} + +/// Pillow-exact BICUBIC resize (RGB8). Horizontal pass then vertical pass with +/// an intermediate u8 buffer, matching `PIL.Image.resize(.., BICUBIC)`. +pub fn resize_bicubic_pil(image: &DynamicImage, out_w: u32, out_h: u32) -> DynamicImage { + let rgb = image.to_rgb8(); + let (in_w, in_h) = rgb.dimensions(); + let (in_w, in_h, out_w_u, out_h_u) = + (in_w as usize, in_h as usize, out_w as usize, out_h as usize); + let horiz = pil_resample_horizontal(rgb.as_raw(), in_h, in_w, out_w_u, 3); + let vert = pil_resample_vertical(&horiz, in_h, out_w_u, out_h_u, 3); + DynamicImage::ImageRgb8( + RgbImage::from_raw(out_w, out_h, vert).expect("pil resize buffer size"), + ) +} + +/// Resize image preserving aspect ratio, fitting within max dimensions. +pub fn resize_to_fit( + image: &DynamicImage, + max_width: u32, + max_height: u32, + filter: FilterType, +) -> DynamicImage { + let (w, h) = image.dimensions(); + let ratio = (max_width as f64 / w as f64).min(max_height as f64 / h as f64); + if ratio >= 1.0 { + return image.clone(); + } + let new_w = ((w as f64 * ratio).round() as u32).max(1); + let new_h = ((h as f64 * ratio).round() as u32).max(1); + resize(image, new_w, new_h, filter) +} + /// Center crop image to specified dimensions. /// /// If the crop size is larger than the image, the image is returned unchanged. diff --git a/crates/multimodal/tests/resize_fingerprint.rs b/crates/multimodal/tests/resize_fingerprint.rs new file mode 100644 index 0000000000..b8ebd5997d --- /dev/null +++ b/crates/multimodal/tests/resize_fingerprint.rs @@ -0,0 +1,68 @@ +//! Bit-identity guard for resize_bicubic_pil. The fingerprints below pin the +//! EXACT byte output of the Pillow-exact BICUBIC resize. Any change to the +//! resize (e.g. parallelization for speed) MUST keep these identical — the +//! resize feeds vision-encoder input, so its output must stay bit-for-bit +//! stable to preserve vLLM/PIL parity (accuracy). +use image::{DynamicImage, RgbImage}; +use llm_multimodal::vision::transforms::resize_bicubic_pil; + +fn make(w: u32, h: u32) -> DynamicImage { + // deterministic, non-trivial structure across all 3 channels + let img = RgbImage::from_fn(w, h, |x, y| { + image::Rgb([ + ((x * 7 + y * 3) % 256) as u8, + ((x * 5 + y * 11) % 256) as u8, + ((x + y * 2) % 256) as u8, + ]) + }); + DynamicImage::ImageRgb8(img) +} + +fn fnv1a(bytes: &[u8]) -> u64 { + let mut h: u64 = 0xcbf2_9ce4_8422_2325; + for &b in bytes { + h ^= b as u64; + h = h.wrapping_mul(0x0000_0100_0000_01b3); + } + h +} + +const CASES: &[(u32, u32, u32, u32)] = &[ + (800, 600, 200, 150), // downscale + (640, 480, 336, 336), // downscale to square + (1280, 960, 512, 384), // downscale large + (259, 194, 280, 196), // slight upscale (real MMBench-ish) + (200, 200, 700, 700), // upscale +]; + +// Captured under the serial implementation; PARALLELIZATION MUST NOT CHANGE THESE. +const EXPECTED: &[u64] = &[ + 0xac15afa8701536c4, + 0x9b76033374b3e1a2, + 0xf21c4a6ac5c20c83, + 0xb170a19da1087feb, + 0xf09a480918a7e2ad, +]; + +#[test] +#[ignore = "capture mode: prints fingerprints"] +fn capture_resize_fingerprints() { + for (iw, ih, ow, oh) in CASES { + let out = resize_bicubic_pil(&make(*iw, *ih), *ow, *oh); + let h = fnv1a(out.to_rgb8().as_raw()); + println!("{iw}x{ih}->{ow}x{oh}: 0x{h:016x}"); + } +} + +#[test] +fn resize_bicubic_pil_bit_identity() { + if EXPECTED.iter().all(|&v| v == 0) { + eprintln!("EXPECTED not yet filled; run capture_resize_fingerprints"); + return; + } + for ((iw, ih, ow, oh), &exp) in CASES.iter().zip(EXPECTED) { + let out = resize_bicubic_pil(&make(*iw, *ih), *ow, *oh); + let got = fnv1a(out.to_rgb8().as_raw()); + assert_eq!(got, exp, "resize fingerprint changed for {iw}x{ih}->{ow}x{oh}"); + } +} From 1165b1dda8d659a3e91dbeedc11894e5ef5ed5d9 Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 07:45:32 -0700 Subject: [PATCH 14/28] feat(mm): libjpeg-turbo decode + Qwen3-VL preprocessing for PIL/HF parity Bring SMG's image path to bit-for-bit parity with the HF/PIL pipeline that vLLM uses, so vision-encoder inputs match and image accuracy is preserved: - jpeg_turbo: minimal libjpeg-turbo (TurboJPEG) FFI for RGB JPEG decode, matching Pillow's chroma-upsampling (the pure-Rust decoder differs by a few levels and shifts embeddings). build.rs links turbojpeg; media.rs routes JPEG decode through it (falling back to the `image` crate otherwise). - qwen_vl_base: Qwen3-VL preprocess (smart_resize, grid/token calc, fused normalize, patchify) producing HF-equivalent pixel_values + image_grid_thw. Pairs with the Pillow-exact BICUBIC resize already in transforms.rs. Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- crates/multimodal/build.rs | 6 + crates/multimodal/src/jpeg_turbo.rs | 111 ++++++++++++++++++ crates/multimodal/src/lib.rs | 1 + crates/multimodal/src/media.rs | 24 +++- .../src/vision/processors/qwen_vl_base.rs | 14 ++- 5 files changed, 146 insertions(+), 10 deletions(-) create mode 100644 crates/multimodal/build.rs create mode 100644 crates/multimodal/src/jpeg_turbo.rs diff --git a/crates/multimodal/build.rs b/crates/multimodal/build.rs new file mode 100644 index 0000000000..f63dbd34fa --- /dev/null +++ b/crates/multimodal/build.rs @@ -0,0 +1,6 @@ +fn main() { + // Link libjpeg-turbo's TurboJPEG API so JPEG decode matches PIL/libjpeg-turbo + // (what vLLM uses) bit-for-bit. Provided by the `libturbojpeg0-dev` package. + println!("cargo:rustc-link-lib=turbojpeg"); + println!("cargo:rustc-link-search=native=/usr/lib/x86_64-linux-gnu"); +} diff --git a/crates/multimodal/src/jpeg_turbo.rs b/crates/multimodal/src/jpeg_turbo.rs new file mode 100644 index 0000000000..6a8ad02b84 --- /dev/null +++ b/crates/multimodal/src/jpeg_turbo.rs @@ -0,0 +1,111 @@ +//! Minimal FFI to libjpeg-turbo's TurboJPEG API for JPEG decode. +//! +//! PIL/Pillow (and therefore vLLM) decode JPEGs with libjpeg-turbo using its +//! default options: accurate (islow) integer IDCT and "fancy" (bilinear) chroma +//! upsampling. The pure-Rust `image`/`zune-jpeg` decoder differs by a few levels +//! per pixel, which the vision encoder amplifies into a large embedding shift, +//! making TokenSpeed's multimodal accuracy diverge from vLLM. Decoding through +//! libjpeg-turbo with the same defaults makes SMG's pixel values match vLLM's. +//! +//! We bind only the three functions needed for RGB decode. Default flags (0) +//! select accurate DCT + fancy upsampling, matching Pillow. +//! +//! This module is the crate's only FFI surface, so it locally overrides the +//! workspace-wide `unsafe_code = "deny"` for the C bindings. +#![allow(unsafe_code)] + +use std::os::raw::{c_int, c_uchar, c_ulong, c_void}; + +use image::{DynamicImage, RgbImage}; + +type TjHandle = *mut c_void; +const TJPF_RGB: c_int = 0; + +#[link(name = "turbojpeg")] +extern "C" { + fn tjInitDecompress() -> TjHandle; + fn tjDecompressHeader3( + handle: TjHandle, + jpeg_buf: *const c_uchar, + jpeg_size: c_ulong, + width: *mut c_int, + height: *mut c_int, + jpeg_subsamp: *mut c_int, + jpeg_colorspace: *mut c_int, + ) -> c_int; + fn tjDecompress2( + handle: TjHandle, + jpeg_buf: *const c_uchar, + jpeg_size: c_ulong, + dst_buf: *mut c_uchar, + width: c_int, + pitch: c_int, + height: c_int, + pixel_format: c_int, + flags: c_int, + ) -> c_int; + fn tjDestroy(handle: TjHandle) -> c_int; +} + +/// True if `bytes` start with the JPEG SOI marker. +pub fn is_jpeg(bytes: &[u8]) -> bool { + bytes.len() >= 3 && bytes[0] == 0xFF && bytes[1] == 0xD8 && bytes[2] == 0xFF +} + +/// Decode a JPEG to an RGB8 `DynamicImage` via libjpeg-turbo (PIL-compatible +/// defaults). Returns `None` on any failure so the caller can fall back to the +/// pure-Rust decoder. +pub fn decode_jpeg_rgb(bytes: &[u8]) -> Option { + if !is_jpeg(bytes) { + return None; + } + // SAFETY: handle is checked for null; buffers are sized from the decoded + // header; the handle is always destroyed before returning. + unsafe { + let handle = tjInitDecompress(); + if handle.is_null() { + return None; + } + let (mut w, mut h, mut subsamp, mut colorspace) = (0_i32, 0_i32, 0_i32, 0_i32); + let hdr = tjDecompressHeader3( + handle, + bytes.as_ptr(), + bytes.len() as c_ulong, + &mut w, + &mut h, + &mut subsamp, + &mut colorspace, + ); + if hdr != 0 || w <= 0 || h <= 0 { + tjDestroy(handle); + return None; + } + let (wu, hu) = (w as usize, h as usize); + // Guard against absurd dimensions before allocating. + let pixels = wu.checked_mul(hu).and_then(|p| p.checked_mul(3)); + let nbytes = match pixels { + Some(n) => n, + None => { + tjDestroy(handle); + return None; + } + }; + let mut buf = vec![0_u8; nbytes]; + let rc = tjDecompress2( + handle, + bytes.as_ptr(), + bytes.len() as c_ulong, + buf.as_mut_ptr(), + w, + 0, // pitch = 0 -> width * pixelsize + h, + TJPF_RGB, + 0, // default flags: accurate IDCT + fancy upsampling (matches Pillow) + ); + tjDestroy(handle); + if rc != 0 { + return None; + } + RgbImage::from_raw(w as u32, h as u32, buf).map(DynamicImage::ImageRgb8) + } +} diff --git a/crates/multimodal/src/lib.rs b/crates/multimodal/src/lib.rs index 2c1a28cf05..7e63f55580 100644 --- a/crates/multimodal/src/lib.rs +++ b/crates/multimodal/src/lib.rs @@ -1,6 +1,7 @@ pub mod error; pub mod hasher; pub mod hub; +pub mod jpeg_turbo; pub mod media; pub mod registry; pub mod tracker; diff --git a/crates/multimodal/src/media.rs b/crates/multimodal/src/media.rs index 19098dc119..f0fda93269 100644 --- a/crates/multimodal/src/media.rs +++ b/crates/multimodal/src/media.rs @@ -329,12 +329,24 @@ impl MediaConnector { ) -> Result, MediaConnectorError> { let hash = crate::hasher::hash_image(&bytes); - let cursor = std::io::Cursor::new(bytes.clone()); - let reader = image::ImageReader::new(cursor).with_guessed_format()?; - - let image = task::spawn_blocking(move || reader.decode()) - .await - .map_err(MediaConnectorError::Blocking)??; + // Decode JPEGs through libjpeg-turbo (PIL-compatible defaults: accurate + // IDCT + fancy upsampling) so pixel values match vLLM bit-for-bit; the + // pure-Rust decoder diverges by a few levels, which the vision encoder + // amplifies into an embedding shift. Non-JPEG inputs and any turbojpeg + // failure fall back to the `image` crate. + let bytes_for_decode = bytes.clone(); + let image = task::spawn_blocking( + move || -> Result { + if let Some(img) = crate::jpeg_turbo::decode_jpeg_rgb(&bytes_for_decode) { + return Ok(img); + } + let cursor = std::io::Cursor::new(bytes_for_decode); + let reader = image::ImageReader::new(cursor).with_guessed_format()?; + Ok(reader.decode()?) + }, + ) + .await + .map_err(MediaConnectorError::Blocking)??; Ok(Arc::new(ImageFrame::new( image, bytes, detail, source, hash, diff --git a/crates/multimodal/src/vision/processors/qwen_vl_base.rs b/crates/multimodal/src/vision/processors/qwen_vl_base.rs index 89416aa762..80de7ee3c7 100644 --- a/crates/multimodal/src/vision/processors/qwen_vl_base.rs +++ b/crates/multimodal/src/vision/processors/qwen_vl_base.rs @@ -23,7 +23,7 @@ use std::borrow::Cow; -use image::{DynamicImage, GenericImageView}; +use image::{imageops::FilterType, DynamicImage, GenericImageView}; use ndarray::{Array2, Array3}; use crate::{ @@ -32,8 +32,8 @@ use crate::{ preprocessor_config::PreProcessorConfig, processor::{ModelSpecificValue, PreprocessedEncoderInputs, VisionPreProcessor}, transforms::{ - pil_to_filter, resize, resize_rgb_bytes, rgb_bytes, to_tensor, to_tensor_and_normalize, - TransformError, + pil_to_filter, resize, resize_bicubic_pil, resize_rgb_bytes, rgb_bytes, to_tensor, + to_tensor_and_normalize, TransformError, }, }, }; @@ -575,7 +575,13 @@ impl VisionPreProcessor for QwenVLProcessorBase { let needs_resize = config.do_resize.unwrap_or(true) && (w != tw32 || h != th32); let resized; let img_ref = if needs_resize { - resized = resize(image, tw32, th32, filter); + // BICUBIC (Qwen default) must match PIL bit-for-bit so encoder + // inputs equal HF/vLLM; other filters keep the SIMD path. + resized = if filter == FilterType::CatmullRom { + resize_bicubic_pil(image, tw32, th32) + } else { + resize(image, tw32, th32, filter) + }; &resized } else { image From 45e949f129c1c4b940316037a7d46c66dadd8a4a Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 07:46:45 -0700 Subject: [PATCH 15/28] perf(mm): parallelize normalize + patchify (bit-identical) Profiling the preprocess on full-resolution images (this config does not downscale) showed the hotspot was normalize + patchify, not resize: 1920x1440 = normalize 23.9ms + patchify 26.2ms (resize skipped). Both stages are elementwise / per-block independent, so band them across threads with std::thread::scope -- identical arithmetic and write order => BIT-IDENTICAL: - deinterleave_rgb_to_planes (normalize): split the pixel range; each output element depends only on its own input byte. 7.5-20x faster. - patchify_into: split the (gt,pr,pc) blocks; each writes a contiguous, deterministic output region of pure copies (memory-bound, ~1.2-1.4x). par_threads gates on output size / rows-per-thread so small images stay serial. preprocess_fingerprint.rs pins the exact f32 encoder_input bytes and asserts the parallel path reproduces the serial output bit-for-bit (accuracy preserved unconditionally). decode_preprocess_bench.rs is the SMG-vs-HF harness. Measured (Qwen3-VL preprocess, 224-core host): 1280x960 15.5->9.0ms, 1920x1440 49.6->28.6ms, 12MP 862->173ms -- now faster than HF on 1280/12MP. Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- .../src/vision/processors/qwen_vl_base.rs | 100 ++++++++++++++---- crates/multimodal/src/vision/transforms.rs | 39 ++++++- .../tests/decode_preprocess_bench.rs | 54 ++++++++++ .../tests/preprocess_fingerprint.rs | 78 ++++++++++++++ 4 files changed, 247 insertions(+), 24 deletions(-) create mode 100644 crates/multimodal/tests/decode_preprocess_bench.rs create mode 100644 crates/multimodal/tests/preprocess_fingerprint.rs diff --git a/crates/multimodal/src/vision/processors/qwen_vl_base.rs b/crates/multimodal/src/vision/processors/qwen_vl_base.rs index 80de7ee3c7..a06ae5aedc 100644 --- a/crates/multimodal/src/vision/processors/qwen_vl_base.rs +++ b/crates/multimodal/src/vision/processors/qwen_vl_base.rs @@ -32,8 +32,8 @@ use crate::{ preprocessor_config::PreProcessorConfig, processor::{ModelSpecificValue, PreprocessedEncoderInputs, VisionPreProcessor}, transforms::{ - pil_to_filter, resize, resize_bicubic_pil, resize_rgb_bytes, rgb_bytes, to_tensor, - to_tensor_and_normalize, TransformError, + par_threads, pil_to_filter, resize, resize_bicubic_pil, resize_rgb_bytes, rgb_bytes, + to_tensor, to_tensor_and_normalize, TransformError, }, }, }; @@ -342,35 +342,89 @@ impl QwenVLProcessorBase { .collect(); let merged_patch = merge_size * patch_size; - let mut out_idx = base_idx; + let pr_blocks = grid_h / merge_size; + let pc_blocks = grid_w / merge_size; + let n_blocks = grid_t * pr_blocks * pc_blocks; + let block_out = merge_size * merge_size * patch_features; + // Each (gt,pr,pc) block writes a contiguous, deterministic output + // region of pure copies, so banding blocks across threads is + // BIT-IDENTICAL. Small grids stay serial. + let region = &mut output[base_idx..base_idx + n_blocks * block_out]; + let nthreads = par_threads(region.len() * 4, n_blocks); + if nthreads <= 1 { + Self::patchify_block_band( + &planes, width, patch_size, merge_size, temporal_patch_size, merged_patch, + pr_blocks, pc_blocks, 0, region, + ); + } else { + let chunk_blocks = n_blocks.div_ceil(nthreads); + let planes_ref = &planes; + std::thread::scope(|s| { + let mut rest = &mut *region; + let mut b0 = 0usize; + while b0 < n_blocks { + let nb = chunk_blocks.min(n_blocks - b0); + let (band, tail) = rest.split_at_mut(nb * block_out); + rest = tail; + let start = b0; + s.spawn(move || { + Self::patchify_block_band( + planes_ref, width, patch_size, merge_size, temporal_patch_size, + merged_patch, pr_blocks, pc_blocks, start, band, + ) + }); + b0 += nb; + } + }); + } - for _gt in 0..grid_t { - for pr in 0..grid_h / merge_size { - for pc in 0..grid_w / merge_size { - let y0 = pr * merged_patch; - let x0 = pc * merged_patch; + Ok(()) + } - for mh in 0..merge_size { - for mw in 0..merge_size { - for plane in &planes { - for _tp in 0..temporal_patch_size { - for py in 0..patch_size { - let row = (y0 + mh * patch_size + py) * width - + x0 - + mw * patch_size; - output[out_idx..out_idx + patch_size] - .copy_from_slice(&plane[row..row + patch_size]); - out_idx += patch_size; - } - } + /// Fill `band` with the patchified output for blocks + /// `[block_start, block_start + band.len()/block_out)` in (gt, pr, pc) + /// row-major order. Pure gather/copy from `planes`; deterministic and + /// independent per block (safe to call concurrently on disjoint bands). + #[allow(clippy::too_many_arguments)] + fn patchify_block_band( + planes: &[&[f32]], + width: usize, + patch_size: usize, + merge_size: usize, + temporal_patch_size: usize, + merged_patch: usize, + pr_blocks: usize, + pc_blocks: usize, + block_start: usize, + band: &mut [f32], + ) { + let block_out = + merge_size * merge_size * planes.len() * temporal_patch_size * patch_size * patch_size; + let per_t = pr_blocks * pc_blocks; + for (bi, chunk) in band.chunks_mut(block_out).enumerate() { + let blk = block_start + bi; + let rem = blk % per_t; + let pr = rem / pc_blocks; + let pc = rem % pc_blocks; + let y0 = pr * merged_patch; + let x0 = pc * merged_patch; + let mut o = 0usize; + for mh in 0..merge_size { + for mw in 0..merge_size { + for plane in planes { + for _tp in 0..temporal_patch_size { + for py in 0..patch_size { + let row = + (y0 + mh * patch_size + py) * width + x0 + mw * patch_size; + chunk[o..o + patch_size] + .copy_from_slice(&plane[row..row + patch_size]); + o += patch_size; } } } } } } - - Ok(()) } /// Patchify a sequence of frame tensors into Qwen's video patch layout. diff --git a/crates/multimodal/src/vision/transforms.rs b/crates/multimodal/src/vision/transforms.rs index 415fdd5d34..66f86a6374 100644 --- a/crates/multimodal/src/vision/transforms.rs +++ b/crates/multimodal/src/vision/transforms.rs @@ -73,6 +73,43 @@ pub fn deinterleave_rgb_to_planes( debug_assert_eq!(pixels, b_plane.len()); debug_assert!(rgb.len() >= pixels * 3); + // Each output element depends only on its own input byte, so banding the + // pixel range across threads is BIT-IDENTICAL (elementwise f32, no + // reduction). Small images stay serial. + let nthreads = par_threads(pixels * 3 * 4, pixels); + if nthreads <= 1 { + deinterleave_contiguous(rgb, r_plane, g_plane, b_plane, scale, bias); + return; + } + let chunk = pixels.div_ceil(nthreads); + let (mut rr, mut gg, mut bb) = (r_plane, g_plane, b_plane); + std::thread::scope(|s| { + let mut p0 = 0usize; + while p0 < pixels { + let n = chunk.min(pixels - p0); + let (rb, rt) = rr.split_at_mut(n); + let (gb, gt) = gg.split_at_mut(n); + let (bbnd, bt) = bb.split_at_mut(n); + rr = rt; + gg = gt; + bb = bt; + let rgb_band = &rgb[p0 * 3..(p0 + n) * 3]; + s.spawn(move || deinterleave_contiguous(rgb_band, rb, gb, bbnd, scale, bias)); + p0 += n; + } + }); +} + +/// Deinterleave a contiguous pixel range (planes/rgb already sliced to the band). +fn deinterleave_contiguous( + rgb: &[u8], + r_plane: &mut [f32], + g_plane: &mut [f32], + b_plane: &mut [f32], + scale: [f32; 3], + bias: [f32; 3], +) { + let pixels = r_plane.len(); let full_blocks = pixels / 8; let remainder = pixels % 8; @@ -383,7 +420,7 @@ fn pil_clip8(v: i64) -> u8 { /// independent fixed-point integer sum, so banding rows over threads yields /// BIT-IDENTICAL output (no shared accumulation, inner sum order unchanged). /// Small images run serial to avoid thread-spawn overhead. -fn par_threads(out_bytes: usize, out_rows: usize) -> usize { +pub(crate) fn par_threads(out_bytes: usize, out_rows: usize) -> usize { const PAR_MIN_BYTES: usize = 1 << 19; // ~512 KiB output; below this, serial const MIN_ROWS_PER_THREAD: usize = 32; // keep enough work per thread const MAX_THREADS: usize = 32; // spawning hundreds of threads costs more than it saves diff --git a/crates/multimodal/tests/decode_preprocess_bench.rs b/crates/multimodal/tests/decode_preprocess_bench.rs new file mode 100644 index 0000000000..cfb613a7e8 --- /dev/null +++ b/crates/multimodal/tests/decode_preprocess_bench.rs @@ -0,0 +1,54 @@ +//! Microbench: SMG (Rust) JPEG decode (libjpeg-turbo) + Qwen3-VL preprocess, +//! on a real image + the real model preprocessor config. Compare against the +//! HF/PIL path that vLLM uses (scripts: bench_hf_preprocess.py). +//! +//! Run: +//! REAL_JPEG=/path/x.jpg PP_CONFIG=/path/preprocessor_config.json \ +//! cargo test -p llm-multimodal --test decode_preprocess_bench -- --ignored --nocapture +use std::time::Instant; + +use llm_multimodal::jpeg_turbo; +use llm_multimodal::vision::{ + preprocessor_config::PreProcessorConfig, processors::Qwen3VLProcessor, VisionPreProcessor, +}; + +#[test] +#[ignore = "perf microbench; needs REAL_JPEG + PP_CONFIG"] +fn bench_decode_preprocess() { + let jpeg_path = std::env::var("REAL_JPEG") + .unwrap_or_else(|_| "/workspace/yechan/runtime_tmp/pp_probe/images/3508.jpg".into()); + let cfg_path = std::env::var("PP_CONFIG").expect("set PP_CONFIG to preprocessor_config.json"); + + let bytes = std::fs::read(&jpeg_path).expect("read jpeg"); + let config = + PreProcessorConfig::from_json(&std::fs::read_to_string(&cfg_path).expect("read config")) + .expect("parse preprocessor config"); + let proc = Qwen3VLProcessor::new(); + + // warmup + let img = jpeg_turbo::decode_jpeg_rgb(&bytes).expect("turbojpeg decode"); + let _ = proc + .preprocess(std::slice::from_ref(&img), &config) + .expect("preprocess"); + + let n_dec = 300usize; + let t0 = Instant::now(); + for _ in 0..n_dec { + let _ = jpeg_turbo::decode_jpeg_rgb(&bytes).unwrap(); + } + let dec_ms = t0.elapsed().as_secs_f64() * 1000.0 / n_dec as f64; + + let n_pp = 200usize; + let t1 = Instant::now(); + for _ in 0..n_pp { + let _ = proc + .preprocess(std::slice::from_ref(&img), &config) + .unwrap(); + } + let pp_ms = t1.elapsed().as_secs_f64() * 1000.0 / n_pp as f64; + + eprintln!("image: {}x{} ({} bytes jpeg)", img.width(), img.height(), bytes.len()); + eprintln!("SMG(Rust) decode (libjpeg-turbo): {dec_ms:.3} ms/img [{n_dec} iters]"); + eprintln!("SMG(Rust) preprocess (Qwen3-VL) : {pp_ms:.3} ms/img [{n_pp} iters]"); + eprintln!("SMG(Rust) decode+preprocess total: {:.3} ms/img", dec_ms + pp_ms); +} diff --git a/crates/multimodal/tests/preprocess_fingerprint.rs b/crates/multimodal/tests/preprocess_fingerprint.rs new file mode 100644 index 0000000000..b898fbd62f --- /dev/null +++ b/crates/multimodal/tests/preprocess_fingerprint.rs @@ -0,0 +1,78 @@ +//! Bit-identity guard for the full Qwen3-VL preprocess (resize + normalize + +//! patchify). Pins the EXACT f32 encoder_input bytes. Any perf change to those +//! stages (parallelization) MUST keep these identical to preserve vLLM/PIL +//! parity (accuracy). +use image::{DynamicImage, RgbImage}; +use llm_multimodal::vision::{ + preprocessor_config::PreProcessorConfig, processors::Qwen3VLProcessor, VisionPreProcessor, +}; + +fn make(w: u32, h: u32) -> DynamicImage { + let img = RgbImage::from_fn(w, h, |x, y| { + image::Rgb([ + ((x * 7 + y * 3) % 256) as u8, + ((x * 5 + y * 11) % 256) as u8, + ((x + y * 2) % 256) as u8, + ]) + }); + DynamicImage::ImageRgb8(img) +} + +fn config() -> PreProcessorConfig { + PreProcessorConfig::from_json( + r#"{"do_resize":true,"do_normalize":true, + "image_mean":[0.48145466,0.4578275,0.40821073], + "image_std":[0.26862954,0.26130258,0.27577711], + "size":{"shortest_edge":3136,"longest_edge":12845056}, + "resample":3}"#, + ) + .unwrap() +} + +fn fnv1a_f32(data: &[f32]) -> u64 { + let mut h: u64 = 0xcbf2_9ce4_8422_2325; + for &v in data { + for &b in v.to_le_bytes().iter() { + h ^= b as u64; + h = h.wrapping_mul(0x0000_0100_0000_01b3); + } + } + h +} + +const CASES: &[(u32, u32)] = &[(560, 420), (840, 560), (1280, 960)]; + +// Captured under serial normalize/patchify; PARALLELIZATION MUST NOT CHANGE THESE. +const EXPECTED: &[u64] = &[ + 0x391ca5deba1ff255, + 0x5bde4728a72eba9d, + 0x617d3e39f58f1c45, +]; + +fn fingerprint(w: u32, h: u32) -> (u64, usize) { + let proc = Qwen3VLProcessor::new(); + let res = proc.preprocess(&[make(w, h)], &config()).unwrap(); + let flat = res.encoder_input_flat(); + (fnv1a_f32(flat.as_ref()), flat.len()) +} + +#[test] +#[ignore = "capture mode"] +fn capture_preprocess_fingerprints() { + for (w, h) in CASES { + let (fp, n) = fingerprint(*w, *h); + println!("{w}x{h}: 0x{fp:016x} (len {n})"); + } +} + +#[test] +fn preprocess_bit_identity() { + if EXPECTED.iter().all(|&v| v == 0) { + eprintln!("EXPECTED not filled; run capture_preprocess_fingerprints"); + return; + } + for ((w, h), &exp) in CASES.iter().zip(EXPECTED) { + let (fp, _) = fingerprint(*w, *h); + assert_eq!(fp, exp, "preprocess fingerprint changed for {w}x{h}"); + } +} From a3981d16cf4d299e431276af8f83263e72959cd1 Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 08:50:04 -0700 Subject: [PATCH 16/28] =?UTF-8?q?fix(mm):=20address=20review=20=E2=80=94?= =?UTF-8?q?=20portable=20turbojpeg=20link,=20SAFETY=20docs,=20require=20RE?= =?UTF-8?q?AL=5FJPEG?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - build.rs: derive the libturbojpeg search dir from the target arch (x86_64 / aarch64 Linux) and only add it when it exists, instead of hardcoding /usr/lib/x86_64-linux-gnu, which broke aarch64/macOS/Windows builds. The link-lib directive is always emitted; other platforms resolve via the default linker search path. (coderabbitai) - jpeg_turbo.rs: expand the SAFETY comment to cover handle lifetime, input buffer validity/no-aliasing, output buffer sizing, and the rc==0 guard. (coderabbitai) - decode_preprocess_bench.rs: require REAL_JPEG explicitly (like PP_CONFIG) instead of a machine-specific fallback path. (coderabbitai) Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- crates/multimodal/build.rs | 23 ++++++++++++++++++- crates/multimodal/src/jpeg_turbo.rs | 15 ++++++++++-- .../tests/decode_preprocess_bench.rs | 3 +-- 3 files changed, 36 insertions(+), 5 deletions(-) diff --git a/crates/multimodal/build.rs b/crates/multimodal/build.rs index f63dbd34fa..fc2daed47e 100644 --- a/crates/multimodal/build.rs +++ b/crates/multimodal/build.rs @@ -2,5 +2,26 @@ fn main() { // Link libjpeg-turbo's TurboJPEG API so JPEG decode matches PIL/libjpeg-turbo // (what vLLM uses) bit-for-bit. Provided by the `libturbojpeg0-dev` package. println!("cargo:rustc-link-lib=turbojpeg"); - println!("cargo:rustc-link-search=native=/usr/lib/x86_64-linux-gnu"); + + // On Debian/Ubuntu multiarch layouts libturbojpeg lives under + // /usr/lib/ rather than a default linker search dir. Derive the + // triplet from the build target (not hardcoded to x86_64) and only add the + // path when it actually exists. Other platforms (macOS/Homebrew, Windows, + // non-multiarch distros) resolve turbojpeg via the default linker search + // path, so we add nothing there and avoid breaking the build. + let arch = std::env::var("CARGO_CFG_TARGET_ARCH").unwrap_or_default(); + let os = std::env::var("CARGO_CFG_TARGET_OS").unwrap_or_default(); + if os == "linux" { + let triplet = match arch.as_str() { + "x86_64" => Some("x86_64-linux-gnu"), + "aarch64" => Some("aarch64-linux-gnu"), + _ => None, + }; + if let Some(triplet) = triplet { + let dir = format!("/usr/lib/{triplet}"); + if std::path::Path::new(&dir).exists() { + println!("cargo:rustc-link-search=native={dir}"); + } + } + } } diff --git a/crates/multimodal/src/jpeg_turbo.rs b/crates/multimodal/src/jpeg_turbo.rs index 6a8ad02b84..f2da2fc413 100644 --- a/crates/multimodal/src/jpeg_turbo.rs +++ b/crates/multimodal/src/jpeg_turbo.rs @@ -59,8 +59,19 @@ pub fn decode_jpeg_rgb(bytes: &[u8]) -> Option { if !is_jpeg(bytes) { return None; } - // SAFETY: handle is checked for null; buffers are sized from the decoded - // header; the handle is always destroyed before returning. + // SAFETY: + // - `tjInitDecompress` is null-checked before any use; every early return + // below calls `tjDestroy(handle)` first, so the handle is always freed + // exactly once and never used after destruction. + // - `bytes` is a live `&[u8]`; `bytes.as_ptr()`/`bytes.len()` describe a + // valid, immutable region for the duration of the FFI calls (libjpeg-turbo + // only reads it). + // - The output `buf` is an owned `Vec` sized to exactly `w*h*3` (checked + // for overflow above) from the header dimensions, and `buf.as_mut_ptr()` + // is the sole alias passed to the decoder; with `pitch=0` (=> `w*3`) and + // `TJPF_RGB` the decoder writes at most `w*h*3` bytes, so no out-of-bounds + // write occurs. We only construct the image after `rc == 0` confirms the + // buffer was fully and successfully written. unsafe { let handle = tjInitDecompress(); if handle.is_null() { diff --git a/crates/multimodal/tests/decode_preprocess_bench.rs b/crates/multimodal/tests/decode_preprocess_bench.rs index cfb613a7e8..42ed1d0605 100644 --- a/crates/multimodal/tests/decode_preprocess_bench.rs +++ b/crates/multimodal/tests/decode_preprocess_bench.rs @@ -15,8 +15,7 @@ use llm_multimodal::vision::{ #[test] #[ignore = "perf microbench; needs REAL_JPEG + PP_CONFIG"] fn bench_decode_preprocess() { - let jpeg_path = std::env::var("REAL_JPEG") - .unwrap_or_else(|_| "/workspace/yechan/runtime_tmp/pp_probe/images/3508.jpg".into()); + let jpeg_path = std::env::var("REAL_JPEG").expect("set REAL_JPEG to a JPEG path"); let cfg_path = std::env::var("PP_CONFIG").expect("set PP_CONFIG to preprocessor_config.json"); let bytes = std::fs::read(&jpeg_path).expect("read jpeg"); From ac18f914232301026bc0fffbc3d58475a09ea472 Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 08:50:04 -0700 Subject: [PATCH 17/28] =?UTF-8?q?fix(grpc):=20address=20review=20=E2=80=94?= =?UTF-8?q?=20secure=20/dev/shm=20files=20+=20cleanup=20on=20assembly=20ab?= =?UTF-8?q?ort?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - proto_wrapper.rs: create TokenSpeed SHM files with create_new(true) (no clobber/symlink in world-writable /dev/shm) and owner-only 0o600 mode (cfg(unix)), for both the writability probe and the payload writer. (coderabbitai, Major) - multimodal.rs / proto_wrapper.rs: assemble_tokenspeed now builds items imperatively and, if any item fails after its encoder input was serialized to SHM, unlinks the already-created /dev/shm segments (prior items + the pending tensor) before returning. Previously the partial TokenSpeedTensor::Shm handles were dropped without reaching the send-path cleanup, leaking files on repeated malformed/unsupported multimodal requests. (chatgpt-codex, P2) Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- model_gateway/src/routers/grpc/multimodal.rs | 113 ++++++++++-------- .../src/routers/grpc/proto_wrapper.rs | 57 +++++++-- 2 files changed, 112 insertions(+), 58 deletions(-) diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index f5852d1be8..51d86a649d 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -39,8 +39,8 @@ use crate::routers::grpc::{ client::GrpcClient, context::WorkerSelection, proto_wrapper::{ - tokenspeed_mm_shm_min_bytes, tokenspeed_mm_tensor_transport_mode, - tokenspeed_shm_dev_writable, write_tokenspeed_shm_with, + cleanup_tokenspeed_items_encoder_shm, tokenspeed_mm_shm_min_bytes, + tokenspeed_mm_tensor_transport_mode, tokenspeed_shm_dev_writable, write_tokenspeed_shm_with, SglangMultimodalData, TensorBytes, TokenSpeedModality, TokenSpeedMultimodalData, TokenSpeedMultimodalItem, TokenSpeedTensor, TrtllmMultimodalData, VllmMultimodalData, }, @@ -1080,57 +1080,70 @@ fn assemble_tokenspeed( }; let item_count = precomputed_multimodal_item_count(&intermediate)?; - let items = (0..item_count) - .map(|item_index| { - let item_encoder_input = encoder_input_for_item( - &intermediate.preprocessed, - &intermediate.field_layouts, + // Build items imperatively so that if any step fails partway we can unlink + // the /dev/shm segments already created for prior items' encoder inputs + // (and this item's, once created). `?`/`collect` would drop those + // `TokenSpeedTensor::Shm` handles without ever reaching the send-path + // cleanup, leaking files until the next sweep. + let mut items: Vec = Vec::with_capacity(item_count); + for item_index in 0..item_count { + let item_encoder_input = match encoder_input_for_item( + &intermediate.preprocessed, + &intermediate.field_layouts, + item_index, + ) { + Ok(value) => value, + Err(error) => { + cleanup_tokenspeed_items_encoder_shm(&items, None); + return Err(error); + } + }; + let encoder_input_started = Instant::now(); + let encoder_input = + serialize_array_as_tokenspeed_tensor(&item_encoder_input, &encoder_input_dtype, shm_enabled); + let encoder_input_serialize_ms = encoder_input_started.elapsed().as_secs_f64() * 1000.0; + let model_specific_started = Instant::now(); + let model_specific_tensors = match serialize_model_specific_for_item( + &intermediate.preprocessed.model_specific, + &intermediate.field_layouts, + item_index, + ) { + Ok(value) => value, + Err(error) => { + // `encoder_input` (possibly SHM) was created for this item but the + // item isn't built; clean it plus all prior items. + cleanup_tokenspeed_items_encoder_shm(&items, Some(&encoder_input)); + return Err(error); + } + }; + let model_specific_serialize_ms = model_specific_started.elapsed().as_secs_f64() * 1000.0; + let mm_placeholders = + placeholders_for_item(item_index, &intermediate.placeholders, &patch_offsets); + let content_hash = content_hash_for_item(intermediate.modality, &intermediate, item_index); + + if log_timing { + info!( + modality = ?modality, item_index, - )?; - let encoder_input_started = Instant::now(); - let encoder_input = serialize_array_as_tokenspeed_tensor( - &item_encoder_input, - &encoder_input_dtype, - shm_enabled, + encoder_input_dtype = %encoder_input.dtype, + encoder_input_bytes = encoder_input.nbytes(), + encoder_input_shape = ?encoder_input.shape, + model_specific_tensor_count = model_specific_tensors.len(), + encoder_input_serialize_ms, + model_specific_serialize_ms, + "smg_mm_timing assemble_tokenspeed_item" ); - let encoder_input_serialize_ms = encoder_input_started.elapsed().as_secs_f64() * 1000.0; - let model_specific_started = Instant::now(); - let model_specific_tensors = serialize_model_specific_for_item( - &intermediate.preprocessed.model_specific, - &intermediate.field_layouts, - item_index, - )?; - let model_specific_serialize_ms = - model_specific_started.elapsed().as_secs_f64() * 1000.0; - let mm_placeholders = - placeholders_for_item(item_index, &intermediate.placeholders, &patch_offsets); - let content_hash = - content_hash_for_item(intermediate.modality, &intermediate, item_index); - - if log_timing { - info!( - modality = ?modality, - item_index, - encoder_input_dtype = %encoder_input.dtype, - encoder_input_bytes = encoder_input.nbytes(), - encoder_input_shape = ?encoder_input.shape, - model_specific_tensor_count = model_specific_tensors.len(), - encoder_input_serialize_ms, - model_specific_serialize_ms, - "smg_mm_timing assemble_tokenspeed_item" - ); - } + } - Ok(TokenSpeedMultimodalItem { - modality, - encoder_input, - model_specific_tensors, - placeholder_token_id: intermediate.placeholder_token_id, - mm_placeholders, - content_hash, - }) - }) - .collect::>>()?; + items.push(TokenSpeedMultimodalItem { + modality, + encoder_input, + model_specific_tensors, + placeholder_token_id: intermediate.placeholder_token_id, + mm_placeholders, + content_hash, + }); + } if log_timing { info!( diff --git a/model_gateway/src/routers/grpc/proto_wrapper.rs b/model_gateway/src/routers/grpc/proto_wrapper.rs index 4e93d8505d..e9d2592a2b 100644 --- a/model_gateway/src/routers/grpc/proto_wrapper.rs +++ b/model_gateway/src/routers/grpc/proto_wrapper.rs @@ -430,10 +430,17 @@ pub fn tokenspeed_shm_dev_writable() -> bool { *WRITABLE.get_or_init(|| { let name = format!("smg-tokenspeed-probe-{}", process::id()); let path = tokenspeed_shm_path(&name); - let ok = OpenOptions::new() - .write(true) - .create(true) - .truncate(true) + // `create_new` (no clobber) + owner-only mode: /dev/shm is world-writable, + // so plain create(truncate) is open to symlink/clobber attacks and the + // file would otherwise inherit umask and be world-readable. + let mut opts = OpenOptions::new(); + opts.write(true).create_new(true); + #[cfg(unix)] + { + use std::os::unix::fs::OpenOptionsExt; + opts.mode(0o600); + } + let ok = opts .open(&path) .and_then(|mut file| file.write_all(b"x")) .is_ok(); @@ -504,10 +511,15 @@ pub fn write_tokenspeed_shm_with( sweep_orphan_tokenspeed_shm_once(); let name = next_tokenspeed_shm_name(); let path = tokenspeed_shm_path(&name); - let file = OpenOptions::new() - .write(true) - .create_new(true) - .open(&path)?; + // create_new (no clobber) + owner-only mode in world-writable /dev/shm. + let mut opts = OpenOptions::new(); + opts.write(true).create_new(true); + #[cfg(unix)] + { + use std::os::unix::fs::OpenOptionsExt; + opts.mode(0o600); + } + let file = opts.open(&path)?; let mut writer = BufWriter::new(file); if let Err(error) = write_fn(&mut writer) { drop(writer); @@ -583,6 +595,35 @@ pub fn cleanup_tokenspeed_shm_handles(handles: &[tokenspeed::ShmHandle]) { } } +/// Unlink the `/dev/shm` segments backing the encoder inputs of intermediate +/// `items` (plus an optional just-built `pending` tensor that hasn't been pushed +/// yet). Used when multimodal assembly aborts partway: the successfully built +/// `TokenSpeedTensor::Shm` segments would otherwise be dropped without their +/// handles ever reaching the send-path cleanup hooks, leaking files until the +/// next process sweep. Only the encoder input uses SHM (model-specific tensors +/// stay inline). MUST run on the error path only — the success path keeps the +/// files alive for the worker and unlinks them after the RPC. +pub(crate) fn cleanup_tokenspeed_items_encoder_shm( + items: &[TokenSpeedMultimodalItem], + pending: Option<&TokenSpeedTensor>, +) { + let mut handles = Vec::new(); + let mut push = |tensor: &TokenSpeedTensor| { + if let TokenSpeedTensorStorage::Shm(handle) = &tensor.storage { + handles.push(handle.clone()); + } + }; + for item in items { + push(&item.encoder_input); + } + if let Some(tensor) = pending { + push(tensor); + } + if !handles.is_empty() { + cleanup_tokenspeed_shm_handles(&handles); + } +} + fn collect_optional_tokenspeed_tensor_shm_handles( tensor: Option<&tokenspeed::TensorData>, handles: &mut Vec, From 4c53004bb978705024b91e7dc175e45a93ffb3f4 Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 08:57:30 -0700 Subject: [PATCH 18/28] =?UTF-8?q?perf(grpc):=20address=20review=20?= =?UTF-8?q?=E2=80=94=20chunk=20u16=20SHM=20conversion=20on=20the=20contigu?= =?UTF-8?q?ous=20path?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit write_array_as_u16's contiguous fast path converted the whole bf16/f16 tensor into a single Vec before writing, so peak memory scaled with encoder-input size. Convert in CHUNK_VALUES-sized blocks (reusing one buffer) like the strided path already does, bounding peak conversion memory to ~512 KiB regardless of tensor size. (coderabbitai) Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- model_gateway/src/routers/grpc/multimodal.rs | 30 ++++++++++++-------- 1 file changed, 18 insertions(+), 12 deletions(-) diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index 51d86a649d..52dfb27314 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -1473,26 +1473,32 @@ fn write_array_as_u16( where F: Fn(f32) -> u16 + Copy, { + // Convert in bounded chunks so peak memory stays at ~CHUNK_VALUES u16s + // regardless of tensor size, on both the contiguous and strided paths. + const CHUNK_VALUES: usize = 256 * 1024; + if let Some(encoder_slice) = encoder_input .as_slice() .or_else(|| encoder_input.as_slice_memory_order()) { - let converted: Vec = encoder_slice.iter().map(|&value| convert(value)).collect(); - #[cfg(target_endian = "little")] - { - return writer.write_all(bytemuck::cast_slice(converted.as_slice())); - } - #[cfg(not(target_endian = "little"))] - { - for value in converted { - writer.write_all(&value.to_le_bytes())?; + let mut converted: Vec = Vec::with_capacity(CHUNK_VALUES.min(encoder_slice.len())); + for chunk in encoder_slice.chunks(CHUNK_VALUES) { + converted.clear(); + converted.extend(chunk.iter().map(|&value| convert(value))); + #[cfg(target_endian = "little")] + { + writer.write_all(bytemuck::cast_slice(converted.as_slice()))?; + } + #[cfg(not(target_endian = "little"))] + { + for value in &converted { + writer.write_all(&value.to_le_bytes())?; + } } - return Ok(()); } + return Ok(()); } - const CHUNK_VALUES: usize = 256 * 1024; - let mut converted = Vec::with_capacity(CHUNK_VALUES); let mut flush = |converted: &mut Vec| -> std::io::Result<()> { if converted.is_empty() { From 020d7b735f654b5295370ae83bb8a72a6194337d Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 09:06:24 -0700 Subject: [PATCH 19/28] fix(mm): load libturbojpeg at runtime (dlopen) instead of linking MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CI failed linking the Go/Python bindings: `cargo:rustc-link-lib=turbojpeg` made libturbojpeg a hard link-time dependency for every consumer of llm-multimodal, but the binding/CI runners don't ship it (`rust-lld: unable to find library -lturbojpeg`). The earlier build.rs path tweak didn't help — the link requirement itself was the problem. Load libturbojpeg via dlopen (libloading) at first use instead: try the runtime soname then the dev/macOS names, resolve the four TurboJPEG symbols, and cache the handle. No build script, no link-time dependency, so the crate and all consumers compile on any platform. Where the library is present (the serving image) decode still goes through it for PIL/vLLM parity; where it's absent decode_jpeg_rgb returns None and the caller falls back to the pure-Rust decoder. Removes build.rs and the #[link]/extern block; adds libloading. Verified the runtime path still decodes (3508.jpg 259x194, ~0.1ms/img). Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- crates/multimodal/Cargo.toml | 1 + crates/multimodal/build.rs | 27 ----- crates/multimodal/src/jpeg_turbo.rs | 154 +++++++++++++++++++--------- 3 files changed, 107 insertions(+), 75 deletions(-) delete mode 100644 crates/multimodal/build.rs diff --git a/crates/multimodal/Cargo.toml b/crates/multimodal/Cargo.toml index f5608cdcc9..f828e4aa53 100644 --- a/crates/multimodal/Cargo.toml +++ b/crates/multimodal/Cargo.toml @@ -25,6 +25,7 @@ hf-hub = { version = "0.5.0", default-features = false, features = ["tokio", "ru bytes = { version = "1.12.0", features = ["serde"] } fast_image_resize = { version = "6.0.0", features = ["image"] } image = { version = "0.25.10", default-features = false, features = ["png", "jpeg", "gif", "bmp", "ico", "tiff", "webp"] } +libloading = "0.8" ndarray = "0.17" once_cell = "1.21.4" opencv = { version = "0.98.2", default-features = false, features = ["clang-runtime", "imgproc", "videoio"], optional = true } diff --git a/crates/multimodal/build.rs b/crates/multimodal/build.rs deleted file mode 100644 index fc2daed47e..0000000000 --- a/crates/multimodal/build.rs +++ /dev/null @@ -1,27 +0,0 @@ -fn main() { - // Link libjpeg-turbo's TurboJPEG API so JPEG decode matches PIL/libjpeg-turbo - // (what vLLM uses) bit-for-bit. Provided by the `libturbojpeg0-dev` package. - println!("cargo:rustc-link-lib=turbojpeg"); - - // On Debian/Ubuntu multiarch layouts libturbojpeg lives under - // /usr/lib/ rather than a default linker search dir. Derive the - // triplet from the build target (not hardcoded to x86_64) and only add the - // path when it actually exists. Other platforms (macOS/Homebrew, Windows, - // non-multiarch distros) resolve turbojpeg via the default linker search - // path, so we add nothing there and avoid breaking the build. - let arch = std::env::var("CARGO_CFG_TARGET_ARCH").unwrap_or_default(); - let os = std::env::var("CARGO_CFG_TARGET_OS").unwrap_or_default(); - if os == "linux" { - let triplet = match arch.as_str() { - "x86_64" => Some("x86_64-linux-gnu"), - "aarch64" => Some("aarch64-linux-gnu"), - _ => None, - }; - if let Some(triplet) = triplet { - let dir = format!("/usr/lib/{triplet}"); - if std::path::Path::new(&dir).exists() { - println!("cargo:rustc-link-search=native={dir}"); - } - } - } -} diff --git a/crates/multimodal/src/jpeg_turbo.rs b/crates/multimodal/src/jpeg_turbo.rs index f2da2fc413..4fedf64d2b 100644 --- a/crates/multimodal/src/jpeg_turbo.rs +++ b/crates/multimodal/src/jpeg_turbo.rs @@ -1,4 +1,4 @@ -//! Minimal FFI to libjpeg-turbo's TurboJPEG API for JPEG decode. +//! Runtime (dlopen) binding to libjpeg-turbo's TurboJPEG API for JPEG decode. //! //! PIL/Pillow (and therefore vLLM) decode JPEGs with libjpeg-turbo using its //! default options: accurate (islow) integer IDCT and "fancy" (bilinear) chroma @@ -7,44 +7,104 @@ //! making TokenSpeed's multimodal accuracy diverge from vLLM. Decoding through //! libjpeg-turbo with the same defaults makes SMG's pixel values match vLLM's. //! -//! We bind only the three functions needed for RGB decode. Default flags (0) -//! select accurate DCT + fancy upsampling, matching Pillow. +//! We load libturbojpeg at RUNTIME via `dlopen` rather than linking it, so the +//! crate (and every consumer — including the Go/Python bindings and CI builds +//! that don't ship libturbojpeg) compiles on any platform with no build script +//! and no link-time dependency. Where the shared library is present (the serving +//! image), decode goes through it for PIL parity; where it's absent, +//! `decode_jpeg_rgb` returns `None` and the caller falls back to the pure-Rust +//! decoder. Default flags (0) select accurate DCT + fancy upsampling, matching +//! Pillow. //! //! This module is the crate's only FFI surface, so it locally overrides the //! workspace-wide `unsafe_code = "deny"` for the C bindings. #![allow(unsafe_code)] use std::os::raw::{c_int, c_uchar, c_ulong, c_void}; +use std::sync::OnceLock; use image::{DynamicImage, RgbImage}; +use libloading::{Library, Symbol}; type TjHandle = *mut c_void; const TJPF_RGB: c_int = 0; -#[link(name = "turbojpeg")] -extern "C" { - fn tjInitDecompress() -> TjHandle; - fn tjDecompressHeader3( - handle: TjHandle, - jpeg_buf: *const c_uchar, - jpeg_size: c_ulong, - width: *mut c_int, - height: *mut c_int, - jpeg_subsamp: *mut c_int, - jpeg_colorspace: *mut c_int, - ) -> c_int; - fn tjDecompress2( - handle: TjHandle, - jpeg_buf: *const c_uchar, - jpeg_size: c_ulong, - dst_buf: *mut c_uchar, - width: c_int, - pitch: c_int, - height: c_int, - pixel_format: c_int, - flags: c_int, - ) -> c_int; - fn tjDestroy(handle: TjHandle) -> c_int; +type TjInitDecompress = unsafe extern "C" fn() -> TjHandle; +type TjDecompressHeader3 = unsafe extern "C" fn( + TjHandle, + *const c_uchar, + c_ulong, + *mut c_int, + *mut c_int, + *mut c_int, + *mut c_int, +) -> c_int; +type TjDecompress2 = unsafe extern "C" fn( + TjHandle, + *const c_uchar, + c_ulong, + *mut c_uchar, + c_int, + c_int, + c_int, + c_int, + c_int, +) -> c_int; +type TjDestroy = unsafe extern "C" fn(TjHandle) -> c_int; + +/// Resolved TurboJPEG entry points. Holds the loaded `Library` so the function +/// pointers stay valid for the process lifetime. +struct TurboJpeg { + _lib: Library, + init: TjInitDecompress, + header: TjDecompressHeader3, + decompress: TjDecompress2, + destroy: TjDestroy, +} + +// The function pointers are plain C entry points with no shared mutable state; +// the library handle is kept alive for the process and never mutated. +unsafe impl Send for TurboJpeg {} +unsafe impl Sync for TurboJpeg {} + +fn load_turbojpeg() -> Option { + // Try the runtime soname first (shipped by the runtime package), then the + // dev symlink and common macOS names. + const CANDIDATES: &[&str] = &[ + "libturbojpeg.so.0", + "libturbojpeg.so", + "libturbojpeg.0.dylib", + "libturbojpeg.dylib", + ]; + // SAFETY: loading a system shared library by name; we only resolve the four + // documented TurboJPEG symbols below and keep the handle for their lifetime. + let lib = CANDIDATES + .iter() + .find_map(|name| unsafe { Library::new(name) }.ok())?; + // SAFETY: each symbol is resolved against the just-loaded library with the + // signature documented by the TurboJPEG API. We copy the bare function + // pointers out (dropping the borrowing `Symbol`s) and keep `lib` alive in + // the returned struct, so the pointers remain valid. + let (init, header, decompress, destroy) = unsafe { + let init: Symbol = lib.get(b"tjInitDecompress\0").ok()?; + let header: Symbol = lib.get(b"tjDecompressHeader3\0").ok()?; + let decompress: Symbol = lib.get(b"tjDecompress2\0").ok()?; + let destroy: Symbol = lib.get(b"tjDestroy\0").ok()?; + (*init, *header, *decompress, *destroy) + }; + Some(TurboJpeg { + _lib: lib, + init, + header, + decompress, + destroy, + }) +} + +/// Process-wide cached TurboJPEG binding, or `None` if the library is absent. +fn turbojpeg() -> Option<&'static TurboJpeg> { + static TJ: OnceLock> = OnceLock::new(); + TJ.get_or_init(load_turbojpeg).as_ref() } /// True if `bytes` start with the JPEG SOI marker. @@ -53,32 +113,31 @@ pub fn is_jpeg(bytes: &[u8]) -> bool { } /// Decode a JPEG to an RGB8 `DynamicImage` via libjpeg-turbo (PIL-compatible -/// defaults). Returns `None` on any failure so the caller can fall back to the -/// pure-Rust decoder. +/// defaults). Returns `None` on any failure — including libturbojpeg not being +/// available at runtime — so the caller can fall back to the pure-Rust decoder. pub fn decode_jpeg_rgb(bytes: &[u8]) -> Option { if !is_jpeg(bytes) { return None; } + let tj = turbojpeg()?; // SAFETY: - // - `tjInitDecompress` is null-checked before any use; every early return - // below calls `tjDestroy(handle)` first, so the handle is always freed + // - `tj.init` returns a handle that is null-checked before use; every early + // return below calls `tj.destroy(handle)` first, so the handle is freed // exactly once and never used after destruction. - // - `bytes` is a live `&[u8]`; `bytes.as_ptr()`/`bytes.len()` describe a - // valid, immutable region for the duration of the FFI calls (libjpeg-turbo - // only reads it). - // - The output `buf` is an owned `Vec` sized to exactly `w*h*3` (checked - // for overflow above) from the header dimensions, and `buf.as_mut_ptr()` - // is the sole alias passed to the decoder; with `pitch=0` (=> `w*3`) and - // `TJPF_RGB` the decoder writes at most `w*h*3` bytes, so no out-of-bounds - // write occurs. We only construct the image after `rc == 0` confirms the - // buffer was fully and successfully written. + // - `bytes` is a live `&[u8]`; its ptr/len describe a valid immutable region + // for the duration of the calls (libjpeg-turbo only reads it). + // - `buf` is an owned `Vec` sized to exactly `w*h*3` (overflow-checked) + // from the header dimensions, and is the sole alias passed to the decoder; + // with `pitch=0` (=> `w*3`) and `TJPF_RGB` the decoder writes at most + // `w*h*3` bytes, so no out-of-bounds write occurs. The image is built only + // after `rc == 0` confirms a successful, complete write. unsafe { - let handle = tjInitDecompress(); + let handle = (tj.init)(); if handle.is_null() { return None; } let (mut w, mut h, mut subsamp, mut colorspace) = (0_i32, 0_i32, 0_i32, 0_i32); - let hdr = tjDecompressHeader3( + let hdr = (tj.header)( handle, bytes.as_ptr(), bytes.len() as c_ulong, @@ -88,21 +147,20 @@ pub fn decode_jpeg_rgb(bytes: &[u8]) -> Option { &mut colorspace, ); if hdr != 0 || w <= 0 || h <= 0 { - tjDestroy(handle); + (tj.destroy)(handle); return None; } let (wu, hu) = (w as usize, h as usize); // Guard against absurd dimensions before allocating. - let pixels = wu.checked_mul(hu).and_then(|p| p.checked_mul(3)); - let nbytes = match pixels { + let nbytes = match wu.checked_mul(hu).and_then(|p| p.checked_mul(3)) { Some(n) => n, None => { - tjDestroy(handle); + (tj.destroy)(handle); return None; } }; let mut buf = vec![0_u8; nbytes]; - let rc = tjDecompress2( + let rc = (tj.decompress)( handle, bytes.as_ptr(), bytes.len() as c_ulong, @@ -113,7 +171,7 @@ pub fn decode_jpeg_rgb(bytes: &[u8]) -> Option { TJPF_RGB, 0, // default flags: accurate IDCT + fancy upsampling (matches Pillow) ); - tjDestroy(handle); + (tj.destroy)(handle); if rc != 0 { return None; } From e026d7a27e5cda61fc58514acda5e0956dba83f0 Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 19:50:55 -0700 Subject: [PATCH 20/28] =?UTF-8?q?refactor(grpc):=20address=20review=20?= =?UTF-8?q?=E2=80=94=20keep=20MM=20transport=20engine-neutral=20(slin1237)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Per review: SMG should stay engine-neutral; only the protocol layer may be engine-specific. - metrics: rename the runtime-baked counters smg_tokenspeed_mm_* -> smg_mm_{tensors,tensor_bytes,shm_write_failures}_total and carry the engine as a `runtime` label (matching the existing runtime-labeled metrics) instead of the metric name. record_mm_tensor / record_mm_shm_write_failure take `runtime`; TokenSpeed call sites pass "tokenspeed". - client.rs: the per-engine build_chat/build_messages dispatch arms no longer inline TokenSpeed SHM-handle collection + error-path cleanup. That engine-specific SHM lifecycle moves into proto_wrapper::finish_tokenspeed_request (the protocol layer), so the dispatch arms are thin like the other engines'. No behavior change (88 grpc tests pass); metric series are renamed + gain a runtime label. Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- model_gateway/src/observability/metrics.rs | 28 ++++----- model_gateway/src/routers/grpc/client.rs | 62 +++++++------------ model_gateway/src/routers/grpc/multimodal.rs | 2 +- .../src/routers/grpc/proto_wrapper.rs | 37 +++++++++-- 4 files changed, 68 insertions(+), 61 deletions(-) diff --git a/model_gateway/src/observability/metrics.rs b/model_gateway/src/observability/metrics.rs index 457ae3d9a1..cdf63e2296 100644 --- a/model_gateway/src/observability/metrics.rs +++ b/model_gateway/src/observability/metrics.rs @@ -389,18 +389,18 @@ pub(crate) fn init_metrics() { ); describe_counter!("smg_db_items_stored", "Total items stored by storage_type"); - // TokenSpeed multimodal tensor transport (shm vs inline) + // Multimodal tensor transport (shm vs inline), labeled by runtime. describe_counter!( - "smg_tokenspeed_mm_tensors_total", - "TokenSpeed multimodal tensors sent, by transport path (shm/inline)" + "smg_mm_tensors_total", + "Multimodal tensors sent, by runtime and transport path (shm/inline)" ); describe_counter!( - "smg_tokenspeed_mm_tensor_bytes_total", - "TokenSpeed multimodal tensor bytes sent, by transport path (shm/inline)" + "smg_mm_tensor_bytes_total", + "Multimodal tensor bytes sent, by runtime and transport path (shm/inline)" ); describe_counter!( - "smg_tokenspeed_mm_shm_write_failures_total", - "TokenSpeed SHM tensor write attempts that failed and fell back to inline" + "smg_mm_shm_write_failures_total", + "SHM tensor write attempts that failed and fell back to inline, by runtime" ); // Layer 0: Tokio runtime self-observability (event-loop canary + sampler). @@ -646,16 +646,16 @@ impl Metrics { .increment(1); } - /// Record one TokenSpeed multimodal tensor sent over `path` ("shm"|"inline"). - pub fn record_tokenspeed_mm_tensor(path: &'static str, nbytes: usize) { - counter!("smg_tokenspeed_mm_tensors_total", "path" => path).increment(1); - counter!("smg_tokenspeed_mm_tensor_bytes_total", "path" => path) + /// Record one multimodal tensor sent over `path` ("shm"|"inline") for `runtime`. + pub fn record_mm_tensor(runtime: &'static str, path: &'static str, nbytes: usize) { + counter!("smg_mm_tensors_total", "runtime" => runtime, "path" => path).increment(1); + counter!("smg_mm_tensor_bytes_total", "runtime" => runtime, "path" => path) .increment(nbytes as u64); } - /// Record a TokenSpeed SHM write that failed and fell back to inline. - pub fn record_tokenspeed_mm_shm_write_failure() { - counter!("smg_tokenspeed_mm_shm_write_failures_total").increment(1); + /// Record a SHM tensor write that failed and fell back to inline, for `runtime`. + pub fn record_mm_shm_write_failure(runtime: &'static str) { + counter!("smg_mm_shm_write_failures_total", "runtime" => runtime).increment(1); } // ======================================================================== diff --git a/model_gateway/src/routers/grpc/client.rs b/model_gateway/src/routers/grpc/client.rs index fae781a141..f9784ace80 100644 --- a/model_gateway/src/routers/grpc/client.rs +++ b/model_gateway/src/routers/grpc/client.rs @@ -15,8 +15,8 @@ use smg_grpc_client::{ use crate::routers::grpc::{ proto_wrapper::{ cleanup_tokenspeed_shm_handles, collect_tokenspeed_generate_request_shm_handles, - collect_tokenspeed_multimodal_inputs_shm_handles, ProtoEmbedComplete, ProtoEmbedRequest, - ProtoGenerateRequest, ProtoStream, + finish_tokenspeed_request, ProtoEmbedComplete, ProtoEmbedRequest, ProtoGenerateRequest, + ProtoStream, }, MultimodalData, }; @@ -530,25 +530,16 @@ impl GrpcClient { MultimodalData::TokenSpeed(data) => data.into_proto(), _ => unreachable!("caller guarantees matching variant"), }); - let shm_handles = tokenspeed_mm - .as_ref() - .map(collect_tokenspeed_multimodal_inputs_shm_handles) - .unwrap_or_default(); - let req = match client.build_generate_request_from_chat( - request_id, - body, - processed_text, - token_ids, - tokenspeed_mm, - options.tool_constraints, - ) { - Ok(req) => req, - Err(error) => { - cleanup_tokenspeed_shm_handles(&shm_handles); - return Err(error); - } - }; - Ok(ProtoGenerateRequest::TokenSpeed(Box::new(req))) + finish_tokenspeed_request(tokenspeed_mm, |mm| { + client.build_generate_request_from_chat( + request_id, + body, + processed_text, + token_ids, + mm, + options.tool_constraints, + ) + }) } } } @@ -630,25 +621,16 @@ impl GrpcClient { MultimodalData::TokenSpeed(data) => data.into_proto(), _ => unreachable!("caller guarantees matching variant"), }); - let shm_handles = tokenspeed_mm - .as_ref() - .map(collect_tokenspeed_multimodal_inputs_shm_handles) - .unwrap_or_default(); - let req = match client.build_generate_request_from_messages( - request_id, - body, - processed_text, - token_ids, - tokenspeed_mm, - options.tool_constraints, - ) { - Ok(req) => req, - Err(error) => { - cleanup_tokenspeed_shm_handles(&shm_handles); - return Err(error); - } - }; - Ok(ProtoGenerateRequest::TokenSpeed(Box::new(req))) + finish_tokenspeed_request(tokenspeed_mm, |mm| { + client.build_generate_request_from_messages( + request_id, + body, + processed_text, + token_ids, + mm, + options.tool_constraints, + ) + }) } } } diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index 52dfb27314..50e8d832a1 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -1412,7 +1412,7 @@ fn serialize_array_as_tokenspeed_tensor( dtype = %dtype, "Failed to write TokenSpeed encoder input directly to SHM; falling back to bytes path" ); - crate::observability::metrics::Metrics::record_tokenspeed_mm_shm_write_failure(); + crate::observability::metrics::Metrics::record_mm_shm_write_failure("tokenspeed"); } } } diff --git a/model_gateway/src/routers/grpc/proto_wrapper.rs b/model_gateway/src/routers/grpc/proto_wrapper.rs index e9d2592a2b..8771702efe 100644 --- a/model_gateway/src/routers/grpc/proto_wrapper.rs +++ b/model_gateway/src/routers/grpc/proto_wrapper.rs @@ -309,7 +309,8 @@ fn tokenspeed_tensor_to_proto(value: TokenSpeedTensor, shm_enabled: bool) -> tok TokenSpeedTensorStorage::Inline(data) => tokenspeed_tensor_payload(data, shm_enabled), // Encoder input already written directly to SHM upstream — meter it here. TokenSpeedTensorStorage::Shm(handle) => { - crate::observability::metrics::Metrics::record_tokenspeed_mm_tensor( + crate::observability::metrics::Metrics::record_mm_tensor( + "tokenspeed", "shm", handle.nbytes as usize, ); @@ -342,7 +343,7 @@ fn tokenspeed_tensor_payload(data: Vec, shm_enabled: bool) -> tokenspeed::te if log_timing { tracing::info!(nbytes, "smg_mm_timing tokenspeed_tensor_payload_inline"); } - Metrics::record_tokenspeed_mm_tensor("inline", nbytes); + Metrics::record_mm_tensor("tokenspeed", "inline", nbytes); return tokenspeed::tensor_data::Payload::Inline(data); } @@ -355,7 +356,7 @@ fn tokenspeed_tensor_payload(data: Vec, shm_enabled: bool) -> tokenspeed::te "smg_mm_timing tokenspeed_tensor_payload_inline_below_threshold" ); } - Metrics::record_tokenspeed_mm_tensor("inline", nbytes); + Metrics::record_mm_tensor("tokenspeed", "inline", nbytes); return tokenspeed::tensor_data::Payload::Inline(data); } @@ -369,7 +370,7 @@ fn tokenspeed_tensor_payload(data: Vec, shm_enabled: bool) -> tokenspeed::te "smg_mm_timing tokenspeed_shm_write" ); } - Metrics::record_tokenspeed_mm_tensor("shm", nbytes); + Metrics::record_mm_tensor("tokenspeed", "shm", nbytes); tokenspeed::tensor_data::Payload::Shm(handle) } Err(error) => { @@ -378,8 +379,8 @@ fn tokenspeed_tensor_payload(data: Vec, shm_enabled: bool) -> tokenspeed::te nbytes, "Failed to create TokenSpeed SHM tensor payload; falling back to inline" ); - Metrics::record_tokenspeed_mm_shm_write_failure(); - Metrics::record_tokenspeed_mm_tensor("inline", nbytes); + Metrics::record_mm_shm_write_failure("tokenspeed"); + Metrics::record_mm_tensor("tokenspeed", "inline", nbytes); tokenspeed::tensor_data::Payload::Inline(data) } } @@ -595,6 +596,30 @@ pub fn cleanup_tokenspeed_shm_handles(handles: &[tokenspeed::ShmHandle]) { } } +/// Build a TokenSpeed proto `GenerateRequest` from the already-converted +/// multimodal proto, unlinking any `/dev/shm` segments it references if `build` +/// fails — so a build error doesn't leak SHM files before the send-path cleanup +/// can run. Keeping this engine-specific SHM lifecycle in the protocol layer +/// lets the per-engine dispatch in `client.rs` stay a thin, neutral wrapper. +pub(crate) fn finish_tokenspeed_request( + tokenspeed_mm: Option, + build: impl FnOnce( + Option, + ) -> Result, +) -> Result { + let shm_handles = tokenspeed_mm + .as_ref() + .map(collect_tokenspeed_multimodal_inputs_shm_handles) + .unwrap_or_default(); + match build(tokenspeed_mm) { + Ok(req) => Ok(ProtoGenerateRequest::TokenSpeed(Box::new(req))), + Err(error) => { + cleanup_tokenspeed_shm_handles(&shm_handles); + Err(error) + } + } +} + /// Unlink the `/dev/shm` segments backing the encoder inputs of intermediate /// `items` (plus an optional just-built `pending` tensor that hasn't been pushed /// yet). Used when multimodal assembly aborts partway: the successfully built From 5eecda6f74247f2b4deab2fc94c31415e05ae5e8 Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 20:08:17 -0700 Subject: [PATCH 21/28] fix(mm): satisfy clippy -D warnings (CI) Clippy (deny warnings) flagged the new multimodal code: - use #[expect(clippy::too_many_arguments, reason=...)] instead of #[allow] for the band/patchify helpers (allow_attributes lint) - par_threads: .min(MAX).max(1) -> .clamp(1, MAX) (manual_clamp) - add `;` to the thread-spawn closures (semicolon_if_nothing_returned) - resize_bicubic_pil: gate the construction expect() with #[expect(clippy::expect_used, reason=...)] (buffer is sized by construction) - bring Metrics into scope at the two MM-metric call sites instead of fully qualifying (proto_wrapper, multimodal) - de-qualify HashMap and drop diagnostic eprintln! in the chat-template e2e test - fingerprint/bench tests: crate-level allow for unwrap/expect/print in tests; iterate to_le_bytes() by value No behavior change. Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- .../src/vision/processors/qwen_vl_base.rs | 7 +++++-- crates/multimodal/src/vision/transforms.rs | 19 ++++++++++++++----- .../tests/decode_preprocess_bench.rs | 6 ++++++ .../tests/preprocess_fingerprint.rs | 8 +++++++- crates/multimodal/tests/resize_fingerprint.rs | 6 ++++++ model_gateway/src/routers/grpc/multimodal.rs | 3 ++- .../src/routers/grpc/proto_wrapper.rs | 7 ++----- .../src/routers/grpc/utils/chat_utils.rs | 4 +--- 8 files changed, 43 insertions(+), 17 deletions(-) diff --git a/crates/multimodal/src/vision/processors/qwen_vl_base.rs b/crates/multimodal/src/vision/processors/qwen_vl_base.rs index a06ae5aedc..53628778a0 100644 --- a/crates/multimodal/src/vision/processors/qwen_vl_base.rs +++ b/crates/multimodal/src/vision/processors/qwen_vl_base.rs @@ -371,7 +371,7 @@ impl QwenVLProcessorBase { Self::patchify_block_band( planes_ref, width, patch_size, merge_size, temporal_patch_size, merged_patch, pr_blocks, pc_blocks, start, band, - ) + ); }); b0 += nb; } @@ -385,7 +385,10 @@ impl QwenVLProcessorBase { /// `[block_start, block_start + band.len()/block_out)` in (gt, pr, pc) /// row-major order. Pure gather/copy from `planes`; deterministic and /// independent per block (safe to call concurrently on disjoint bands). - #[allow(clippy::too_many_arguments)] + #[expect( + clippy::too_many_arguments, + reason = "block-band patchifier: planes + grid dims + output band" + )] fn patchify_block_band( planes: &[&[f32]], width: usize, diff --git a/crates/multimodal/src/vision/transforms.rs b/crates/multimodal/src/vision/transforms.rs index 66f86a6374..b3104c8e78 100644 --- a/crates/multimodal/src/vision/transforms.rs +++ b/crates/multimodal/src/vision/transforms.rs @@ -432,14 +432,16 @@ pub(crate) fn par_threads(out_bytes: usize, out_rows: usize) -> usize { .unwrap_or(1); (out_rows / MIN_ROWS_PER_THREAD) .min(avail) - .min(MAX_THREADS) - .max(1) + .clamp(1, MAX_THREADS) } /// Process output rows `[oy0, oy0 + out_band.len()/row_out)` of the horizontal /// pass into `out_band`. Horizontal pass preserves row count, so output row i /// reads input row `oy0 + i`. -#[allow(clippy::too_many_arguments)] +#[expect( + clippy::too_many_arguments, + reason = "row-band resampler: precomputed coeffs + dims + output band" +)] fn pil_h_band( src: &[u8], bounds: &[(usize, usize)], @@ -497,7 +499,7 @@ fn pil_resample_horizontal( rest = tail; let start = oy0; s.spawn(move || { - pil_h_band(src, b, k, half, in_w, out_w, channels, start, band) + pil_h_band(src, b, k, half, in_w, out_w, channels, start, band); }); oy0 += n; } @@ -508,7 +510,10 @@ fn pil_resample_horizontal( /// Process output rows `[oy0, oy0 + out_band.len()/row_out)` of the vertical /// pass into `out_band`. -#[allow(clippy::too_many_arguments)] +#[expect( + clippy::too_many_arguments, + reason = "row-band resampler: precomputed coeffs + dims + output band" +)] fn pil_v_band( src: &[u8], bounds: &[(usize, usize)], @@ -579,6 +584,10 @@ pub fn resize_bicubic_pil(image: &DynamicImage, out_w: u32, out_h: u32) -> Dynam (in_w as usize, in_h as usize, out_w as usize, out_h as usize); let horiz = pil_resample_horizontal(rgb.as_raw(), in_h, in_w, out_w_u, 3); let vert = pil_resample_vertical(&horiz, in_h, out_w_u, out_h_u, 3); + #[expect( + clippy::expect_used, + reason = "vert is exactly out_w*out_h*3 bytes by construction" + )] DynamicImage::ImageRgb8( RgbImage::from_raw(out_w, out_h, vert).expect("pil resize buffer size"), ) diff --git a/crates/multimodal/tests/decode_preprocess_bench.rs b/crates/multimodal/tests/decode_preprocess_bench.rs index 42ed1d0605..7188a8ed39 100644 --- a/crates/multimodal/tests/decode_preprocess_bench.rs +++ b/crates/multimodal/tests/decode_preprocess_bench.rs @@ -5,6 +5,12 @@ //! Run: //! REAL_JPEG=/path/x.jpg PP_CONFIG=/path/preprocessor_config.json \ //! cargo test -p llm-multimodal --test decode_preprocess_bench -- --ignored --nocapture +#![allow( + clippy::unwrap_used, + clippy::expect_used, + clippy::print_stdout, + clippy::print_stderr +)] use std::time::Instant; use llm_multimodal::jpeg_turbo; diff --git a/crates/multimodal/tests/preprocess_fingerprint.rs b/crates/multimodal/tests/preprocess_fingerprint.rs index b898fbd62f..babb0d9703 100644 --- a/crates/multimodal/tests/preprocess_fingerprint.rs +++ b/crates/multimodal/tests/preprocess_fingerprint.rs @@ -2,6 +2,12 @@ //! patchify). Pins the EXACT f32 encoder_input bytes. Any perf change to those //! stages (parallelization) MUST keep these identical to preserve vLLM/PIL //! parity (accuracy). +#![allow( + clippy::unwrap_used, + clippy::expect_used, + clippy::print_stdout, + clippy::print_stderr +)] use image::{DynamicImage, RgbImage}; use llm_multimodal::vision::{ preprocessor_config::PreProcessorConfig, processors::Qwen3VLProcessor, VisionPreProcessor, @@ -32,7 +38,7 @@ fn config() -> PreProcessorConfig { fn fnv1a_f32(data: &[f32]) -> u64 { let mut h: u64 = 0xcbf2_9ce4_8422_2325; for &v in data { - for &b in v.to_le_bytes().iter() { + for b in v.to_le_bytes() { h ^= b as u64; h = h.wrapping_mul(0x0000_0100_0000_01b3); } diff --git a/crates/multimodal/tests/resize_fingerprint.rs b/crates/multimodal/tests/resize_fingerprint.rs index b8ebd5997d..f906879458 100644 --- a/crates/multimodal/tests/resize_fingerprint.rs +++ b/crates/multimodal/tests/resize_fingerprint.rs @@ -3,6 +3,12 @@ //! resize (e.g. parallelization for speed) MUST keep these identical — the //! resize feeds vision-encoder input, so its output must stay bit-for-bit //! stable to preserve vLLM/PIL parity (accuracy). +#![allow( + clippy::unwrap_used, + clippy::expect_used, + clippy::print_stdout, + clippy::print_stderr +)] use image::{DynamicImage, RgbImage}; use llm_multimodal::vision::transforms::resize_bicubic_pil; diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index 50e8d832a1..628a9b0fb2 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -1406,13 +1406,14 @@ fn serialize_array_as_tokenspeed_tensor( return TokenSpeedTensor::shm(handle, shape, dtype); } Err(error) => { + use crate::observability::metrics::Metrics; warn!( ?error, nbytes, dtype = %dtype, "Failed to write TokenSpeed encoder input directly to SHM; falling back to bytes path" ); - crate::observability::metrics::Metrics::record_mm_shm_write_failure("tokenspeed"); + Metrics::record_mm_shm_write_failure("tokenspeed"); } } } diff --git a/model_gateway/src/routers/grpc/proto_wrapper.rs b/model_gateway/src/routers/grpc/proto_wrapper.rs index 8771702efe..171d9fd0dd 100644 --- a/model_gateway/src/routers/grpc/proto_wrapper.rs +++ b/model_gateway/src/routers/grpc/proto_wrapper.rs @@ -299,6 +299,7 @@ impl TokenSpeedMultimodalItem { } fn tokenspeed_tensor_to_proto(value: TokenSpeedTensor, shm_enabled: bool) -> tokenspeed::TensorData { + use crate::observability::metrics::Metrics; let TokenSpeedTensor { storage, shape, @@ -309,11 +310,7 @@ fn tokenspeed_tensor_to_proto(value: TokenSpeedTensor, shm_enabled: bool) -> tok TokenSpeedTensorStorage::Inline(data) => tokenspeed_tensor_payload(data, shm_enabled), // Encoder input already written directly to SHM upstream — meter it here. TokenSpeedTensorStorage::Shm(handle) => { - crate::observability::metrics::Metrics::record_mm_tensor( - "tokenspeed", - "shm", - handle.nbytes as usize, - ); + Metrics::record_mm_tensor("tokenspeed", "shm", handle.nbytes as usize); tokenspeed::tensor_data::Payload::Shm(handle) } }; diff --git a/model_gateway/src/routers/grpc/utils/chat_utils.rs b/model_gateway/src/routers/grpc/utils/chat_utils.rs index cc92a35531..d5805bb965 100644 --- a/model_gateway/src/routers/grpc/utils/chat_utils.rs +++ b/model_gateway/src/routers/grpc/utils/chat_utils.rs @@ -1011,7 +1011,6 @@ mod tests { }; let template = std::fs::read_to_string(&path).expect("read template"); let format = detect_chat_template_content_format(&template); - eprintln!("detected content format: {format}"); let messages = vec![ChatMessage::User { content: MessageContent::Parts(vec![ @@ -1033,7 +1032,7 @@ mod tests { let transformed = process_content_format(&messages, format, Some("<|image_pad|>")).unwrap(); - let mut kwargs = std::collections::HashMap::new(); + let mut kwargs = HashMap::new(); kwargs.insert("enable_thinking".to_string(), json!(false)); let params = ChatTemplateParams { add_generation_prompt: true, @@ -1051,7 +1050,6 @@ mod tests { let qpos = rendered .find("Question:") .expect("rendered prompt has the question"); - eprintln!("vision_start@{vstart} question@{qpos}"); assert!( vstart < qpos, "image must precede the question (vstart={vstart}, qpos={qpos}).\n--- rendered ---\n{rendered}" From fdee8fc994cd98746459fa0f925c330c8331188f Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 20:10:58 -0700 Subject: [PATCH 22/28] style: cargo +nightly fmt Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- crates/multimodal/src/jpeg_turbo.rs | 6 +++-- .../src/vision/processors/qwen_vl_base.rs | 24 +++++++++++++++---- crates/multimodal/src/vision/transforms.rs | 8 +++---- .../tests/decode_preprocess_bench.rs | 20 ++++++++++++---- .../tests/preprocess_fingerprint.rs | 6 +---- crates/multimodal/tests/resize_fingerprint.rs | 15 +++++++----- model_gateway/src/routers/grpc/multimodal.rs | 14 +++++++---- .../src/routers/grpc/proto_wrapper.rs | 5 +++- .../src/routers/grpc/utils/chat_utils.rs | 10 ++++---- 9 files changed, 71 insertions(+), 37 deletions(-) diff --git a/crates/multimodal/src/jpeg_turbo.rs b/crates/multimodal/src/jpeg_turbo.rs index 4fedf64d2b..b24c53b972 100644 --- a/crates/multimodal/src/jpeg_turbo.rs +++ b/crates/multimodal/src/jpeg_turbo.rs @@ -20,8 +20,10 @@ //! workspace-wide `unsafe_code = "deny"` for the C bindings. #![allow(unsafe_code)] -use std::os::raw::{c_int, c_uchar, c_ulong, c_void}; -use std::sync::OnceLock; +use std::{ + os::raw::{c_int, c_uchar, c_ulong, c_void}, + sync::OnceLock, +}; use image::{DynamicImage, RgbImage}; use libloading::{Library, Symbol}; diff --git a/crates/multimodal/src/vision/processors/qwen_vl_base.rs b/crates/multimodal/src/vision/processors/qwen_vl_base.rs index 53628778a0..14773e803e 100644 --- a/crates/multimodal/src/vision/processors/qwen_vl_base.rs +++ b/crates/multimodal/src/vision/processors/qwen_vl_base.rs @@ -353,8 +353,16 @@ impl QwenVLProcessorBase { let nthreads = par_threads(region.len() * 4, n_blocks); if nthreads <= 1 { Self::patchify_block_band( - &planes, width, patch_size, merge_size, temporal_patch_size, merged_patch, - pr_blocks, pc_blocks, 0, region, + &planes, + width, + patch_size, + merge_size, + temporal_patch_size, + merged_patch, + pr_blocks, + pc_blocks, + 0, + region, ); } else { let chunk_blocks = n_blocks.div_ceil(nthreads); @@ -369,8 +377,16 @@ impl QwenVLProcessorBase { let start = b0; s.spawn(move || { Self::patchify_block_band( - planes_ref, width, patch_size, merge_size, temporal_patch_size, - merged_patch, pr_blocks, pc_blocks, start, band, + planes_ref, + width, + patch_size, + merge_size, + temporal_patch_size, + merged_patch, + pr_blocks, + pc_blocks, + start, + band, ); }); b0 += nb; diff --git a/crates/multimodal/src/vision/transforms.rs b/crates/multimodal/src/vision/transforms.rs index b3104c8e78..8c80f0300b 100644 --- a/crates/multimodal/src/vision/transforms.rs +++ b/crates/multimodal/src/vision/transforms.rs @@ -486,7 +486,9 @@ fn pil_resample_horizontal( let mut out = vec![0_u8; rows * row_out]; let nthreads = par_threads(out.len(), rows); if nthreads <= 1 { - pil_h_band(src, &bounds, &kernels, half, in_w, out_w, channels, 0, &mut out); + pil_h_band( + src, &bounds, &kernels, half, in_w, out_w, channels, 0, &mut out, + ); } else { let chunk_rows = rows.div_ceil(nthreads); std::thread::scope(|s| { @@ -588,9 +590,7 @@ pub fn resize_bicubic_pil(image: &DynamicImage, out_w: u32, out_h: u32) -> Dynam clippy::expect_used, reason = "vert is exactly out_w*out_h*3 bytes by construction" )] - DynamicImage::ImageRgb8( - RgbImage::from_raw(out_w, out_h, vert).expect("pil resize buffer size"), - ) + DynamicImage::ImageRgb8(RgbImage::from_raw(out_w, out_h, vert).expect("pil resize buffer size")) } /// Resize image preserving aspect ratio, fitting within max dimensions. diff --git a/crates/multimodal/tests/decode_preprocess_bench.rs b/crates/multimodal/tests/decode_preprocess_bench.rs index 7188a8ed39..64e8fda6b7 100644 --- a/crates/multimodal/tests/decode_preprocess_bench.rs +++ b/crates/multimodal/tests/decode_preprocess_bench.rs @@ -13,9 +13,11 @@ )] use std::time::Instant; -use llm_multimodal::jpeg_turbo; -use llm_multimodal::vision::{ - preprocessor_config::PreProcessorConfig, processors::Qwen3VLProcessor, VisionPreProcessor, +use llm_multimodal::{ + jpeg_turbo, + vision::{ + preprocessor_config::PreProcessorConfig, processors::Qwen3VLProcessor, VisionPreProcessor, + }, }; #[test] @@ -52,8 +54,16 @@ fn bench_decode_preprocess() { } let pp_ms = t1.elapsed().as_secs_f64() * 1000.0 / n_pp as f64; - eprintln!("image: {}x{} ({} bytes jpeg)", img.width(), img.height(), bytes.len()); + eprintln!( + "image: {}x{} ({} bytes jpeg)", + img.width(), + img.height(), + bytes.len() + ); eprintln!("SMG(Rust) decode (libjpeg-turbo): {dec_ms:.3} ms/img [{n_dec} iters]"); eprintln!("SMG(Rust) preprocess (Qwen3-VL) : {pp_ms:.3} ms/img [{n_pp} iters]"); - eprintln!("SMG(Rust) decode+preprocess total: {:.3} ms/img", dec_ms + pp_ms); + eprintln!( + "SMG(Rust) decode+preprocess total: {:.3} ms/img", + dec_ms + pp_ms + ); } diff --git a/crates/multimodal/tests/preprocess_fingerprint.rs b/crates/multimodal/tests/preprocess_fingerprint.rs index babb0d9703..85c84b7f72 100644 --- a/crates/multimodal/tests/preprocess_fingerprint.rs +++ b/crates/multimodal/tests/preprocess_fingerprint.rs @@ -49,11 +49,7 @@ fn fnv1a_f32(data: &[f32]) -> u64 { const CASES: &[(u32, u32)] = &[(560, 420), (840, 560), (1280, 960)]; // Captured under serial normalize/patchify; PARALLELIZATION MUST NOT CHANGE THESE. -const EXPECTED: &[u64] = &[ - 0x391ca5deba1ff255, - 0x5bde4728a72eba9d, - 0x617d3e39f58f1c45, -]; +const EXPECTED: &[u64] = &[0x391ca5deba1ff255, 0x5bde4728a72eba9d, 0x617d3e39f58f1c45]; fn fingerprint(w: u32, h: u32) -> (u64, usize) { let proc = Qwen3VLProcessor::new(); diff --git a/crates/multimodal/tests/resize_fingerprint.rs b/crates/multimodal/tests/resize_fingerprint.rs index f906879458..f1ea416f70 100644 --- a/crates/multimodal/tests/resize_fingerprint.rs +++ b/crates/multimodal/tests/resize_fingerprint.rs @@ -34,11 +34,11 @@ fn fnv1a(bytes: &[u8]) -> u64 { } const CASES: &[(u32, u32, u32, u32)] = &[ - (800, 600, 200, 150), // downscale - (640, 480, 336, 336), // downscale to square - (1280, 960, 512, 384), // downscale large - (259, 194, 280, 196), // slight upscale (real MMBench-ish) - (200, 200, 700, 700), // upscale + (800, 600, 200, 150), // downscale + (640, 480, 336, 336), // downscale to square + (1280, 960, 512, 384), // downscale large + (259, 194, 280, 196), // slight upscale (real MMBench-ish) + (200, 200, 700, 700), // upscale ]; // Captured under the serial implementation; PARALLELIZATION MUST NOT CHANGE THESE. @@ -69,6 +69,9 @@ fn resize_bicubic_pil_bit_identity() { for ((iw, ih, ow, oh), &exp) in CASES.iter().zip(EXPECTED) { let out = resize_bicubic_pil(&make(*iw, *ih), *ow, *oh); let got = fnv1a(out.to_rgb8().as_raw()); - assert_eq!(got, exp, "resize fingerprint changed for {iw}x{ih}->{ow}x{oh}"); + assert_eq!( + got, exp, + "resize fingerprint changed for {iw}x{ih}->{ow}x{oh}" + ); } } diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index 628a9b0fb2..7185288fc2 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -40,9 +40,10 @@ use crate::routers::grpc::{ context::WorkerSelection, proto_wrapper::{ cleanup_tokenspeed_items_encoder_shm, tokenspeed_mm_shm_min_bytes, - tokenspeed_mm_tensor_transport_mode, tokenspeed_shm_dev_writable, write_tokenspeed_shm_with, - SglangMultimodalData, TensorBytes, TokenSpeedModality, TokenSpeedMultimodalData, - TokenSpeedMultimodalItem, TokenSpeedTensor, TrtllmMultimodalData, VllmMultimodalData, + tokenspeed_mm_tensor_transport_mode, tokenspeed_shm_dev_writable, + write_tokenspeed_shm_with, SglangMultimodalData, TensorBytes, TokenSpeedModality, + TokenSpeedMultimodalData, TokenSpeedMultimodalItem, TokenSpeedTensor, TrtllmMultimodalData, + VllmMultimodalData, }, MultimodalData, }; @@ -1099,8 +1100,11 @@ fn assemble_tokenspeed( } }; let encoder_input_started = Instant::now(); - let encoder_input = - serialize_array_as_tokenspeed_tensor(&item_encoder_input, &encoder_input_dtype, shm_enabled); + let encoder_input = serialize_array_as_tokenspeed_tensor( + &item_encoder_input, + &encoder_input_dtype, + shm_enabled, + ); let encoder_input_serialize_ms = encoder_input_started.elapsed().as_secs_f64() * 1000.0; let model_specific_started = Instant::now(); let model_specific_tensors = match serialize_model_specific_for_item( diff --git a/model_gateway/src/routers/grpc/proto_wrapper.rs b/model_gateway/src/routers/grpc/proto_wrapper.rs index 171d9fd0dd..9bc3f1a6c1 100644 --- a/model_gateway/src/routers/grpc/proto_wrapper.rs +++ b/model_gateway/src/routers/grpc/proto_wrapper.rs @@ -298,7 +298,10 @@ impl TokenSpeedMultimodalItem { } } -fn tokenspeed_tensor_to_proto(value: TokenSpeedTensor, shm_enabled: bool) -> tokenspeed::TensorData { +fn tokenspeed_tensor_to_proto( + value: TokenSpeedTensor, + shm_enabled: bool, +) -> tokenspeed::TensorData { use crate::observability::metrics::Metrics; let TokenSpeedTensor { storage, diff --git a/model_gateway/src/routers/grpc/utils/chat_utils.rs b/model_gateway/src/routers/grpc/utils/chat_utils.rs index d5805bb965..d327483bad 100644 --- a/model_gateway/src/routers/grpc/utils/chat_utils.rs +++ b/model_gateway/src/routers/grpc/utils/chat_utils.rs @@ -236,7 +236,9 @@ fn transform_content_field( let mut media_parts: Vec = Vec::new(); let mut text_parts: Vec = Vec::new(); for part in content_array { - let Some(obj) = part.as_object() else { continue }; + let Some(obj) = part.as_object() else { + continue; + }; match obj.get("type").and_then(|t| t.as_str()) { Some("text") => { if let Some(t) = obj.get("text").and_then(|t| t.as_str()) { @@ -253,8 +255,7 @@ fn transform_content_field( } if !media_parts.is_empty() || !text_parts.is_empty() { - let ordered: Vec = - media_parts.into_iter().chain(text_parts).collect(); + let ordered: Vec = media_parts.into_iter().chain(text_parts).collect(); *content_value = Value::String(ordered.join("\n")); } } @@ -1029,8 +1030,7 @@ mod tests { name: None, }]; - let transformed = - process_content_format(&messages, format, Some("<|image_pad|>")).unwrap(); + let transformed = process_content_format(&messages, format, Some("<|image_pad|>")).unwrap(); let mut kwargs = HashMap::new(); kwargs.insert("enable_thinking".to_string(), json!(false)); From 71194e0a2def6b887d7841cc4e608db70ec205f5 Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 20:48:19 -0700 Subject: [PATCH 23/28] =?UTF-8?q?fix(mm):=20address=20PR=20review=20?= =?UTF-8?q?=E2=80=94=20PIL-exact=20video=20resize,=20safer=20SHM=20localit?= =?UTF-8?q?y?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - qwen_vl: route default-bicubic video frame resizing through the PIL-exact path (resize_bicubic_pil / new resize_bicubic_pil_rgb) in both preprocess_video and preprocess_video_rgb, matching the image path so video encoder inputs equal HF/vLLM bit-for-bit instead of diverging to the SIMD CatmullRom resizer. - grpc: do not infer a shared /dev/shm from TCP loopback. auto-mode SHM now requires a unix-domain-socket worker (proven same-host) or an explicit SMG_TOKENSPEED_SHM_ASSUME_LOOPBACK_SHARED operator assertion; loopback alone falls back to inline. Explicit shm mode is unchanged. - grpc: SHM cleanup only unlinks names carrying the smg-tokenspeed- prefix this transport creates, never arbitrary top-level /dev/shm entries. Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- .../src/vision/processors/qwen_vl_base.rs | 42 +++++++++++----- crates/multimodal/src/vision/transforms.rs | 35 ++++++++++++++ model_gateway/src/routers/grpc/multimodal.rs | 48 +++++++++++++++---- .../src/routers/grpc/proto_wrapper.rs | 18 ++++++- 4 files changed, 123 insertions(+), 20 deletions(-) diff --git a/crates/multimodal/src/vision/processors/qwen_vl_base.rs b/crates/multimodal/src/vision/processors/qwen_vl_base.rs index 14773e803e..c7c1c51243 100644 --- a/crates/multimodal/src/vision/processors/qwen_vl_base.rs +++ b/crates/multimodal/src/vision/processors/qwen_vl_base.rs @@ -32,8 +32,8 @@ use crate::{ preprocessor_config::PreProcessorConfig, processor::{ModelSpecificValue, PreprocessedEncoderInputs, VisionPreProcessor}, transforms::{ - par_threads, pil_to_filter, resize, resize_bicubic_pil, resize_rgb_bytes, rgb_bytes, - to_tensor, to_tensor_and_normalize, TransformError, + par_threads, pil_to_filter, resize, resize_bicubic_pil, resize_bicubic_pil_rgb, + resize_rgb_bytes, rgb_bytes, to_tensor, to_tensor_and_normalize, TransformError, }, }, }; @@ -762,7 +762,14 @@ impl VisionPreProcessor for QwenVLProcessorBase { let needs_resize = config.do_resize.unwrap_or(true) && (frame.width() != tw32 || frame.height() != th32); if needs_resize { - let resized = resize(frame, tw32, th32, filter); + // BICUBIC (Qwen default) must match PIL bit-for-bit so video + // encoder inputs equal HF/vLLM, same as the image path; other + // filters keep the SIMD resizer. + let resized = if filter == FilterType::CatmullRom { + resize_bicubic_pil(frame, tw32, th32) + } else { + resize(frame, tw32, th32, filter) + }; let (width, height, data) = rgb_bytes(&resized); frame_rgbs.push(VideoFrameRgb { width, @@ -898,14 +905,27 @@ impl VisionPreProcessor for QwenVLProcessorBase { let frame = frames[idx]; let needs_resize = do_resize && (frame.width != tw32 || frame.height != th32); if needs_resize { - let resized = resize_rgb_bytes( - frame.data, - frame.width, - frame.height, - tw32, - th32, - filter, - )?; + // BICUBIC (Qwen default) must match PIL bit-for-bit so video + // encoder inputs equal HF/vLLM, same as the image path; other + // filters keep the SIMD resizer. + let resized = if filter == FilterType::CatmullRom { + resize_bicubic_pil_rgb( + frame.data, + frame.width, + frame.height, + tw32, + th32, + )? + } else { + resize_rgb_bytes( + frame.data, + frame.width, + frame.height, + tw32, + th32, + filter, + )? + }; frame_rgbs.push(VideoFrameRgb { width: tw32 as usize, height: th32 as usize, diff --git a/crates/multimodal/src/vision/transforms.rs b/crates/multimodal/src/vision/transforms.rs index 8c80f0300b..f8063fc7c7 100644 --- a/crates/multimodal/src/vision/transforms.rs +++ b/crates/multimodal/src/vision/transforms.rs @@ -593,6 +593,41 @@ pub fn resize_bicubic_pil(image: &DynamicImage, out_w: u32, out_h: u32) -> Dynam DynamicImage::ImageRgb8(RgbImage::from_raw(out_w, out_h, vert).expect("pil resize buffer size")) } +/// PIL-exact bicubic resize over borrowed interleaved RGB bytes. +/// +/// Byte-for-byte equivalent of [`resize_bicubic_pil`] but for the raw-RGB video +/// frame path (`preprocess_video_rgb`), so default-bicubic video frames match +/// HF/vLLM the same way images do. Returns an `RgbImage` to drop straight into +/// the existing [`resize_rgb_bytes`] call sites. +pub fn resize_bicubic_pil_rgb( + data: &[u8], + width: u32, + height: u32, + out_w: u32, + out_h: u32, +) -> Result { + let (in_w, in_h, out_w_u, out_h_u) = ( + width as usize, + height as usize, + out_w as usize, + out_h as usize, + ); + let expected = in_w.saturating_mul(in_h).saturating_mul(3); + if data.len() != expected { + return Err(TransformError::ShapeError(format!( + "PIL bicubic RGB source has {} bytes, expected {expected} for {width}x{height}", + data.len() + ))); + } + let horiz = pil_resample_horizontal(data, in_h, in_w, out_w_u, 3); + let vert = pil_resample_vertical(&horiz, in_h, out_w_u, out_h_u, 3); + RgbImage::from_raw(out_w, out_h, vert).ok_or_else(|| { + TransformError::ShapeError(format!( + "failed to build PIL bicubic RGB image for {out_w}x{out_h}" + )) + }) +} + /// Resize image preserving aspect ratio, fitting within max dimensions. pub fn resize_to_fit( image: &DynamicImage, diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index 7185288fc2..941fd2555e 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -1644,15 +1644,15 @@ fn tokenspeed_encoder_input_dtype_from_worker(workers: Option<&WorkerSelection>) /// Resolve whether large multimodal tensors should use the SHM transport for /// this request. `shm` = always (legacy explicit opt-in); `auto` = only when the -/// worker is local and therefore shares SMG's `/dev/shm`; anything else -/// (including unset or `inline`) keeps the inline gRPC path. +/// worker is known to share SMG's `/dev/shm`; anything else (including unset or +/// `inline`) keeps the inline gRPC path. fn resolve_tokenspeed_shm_enabled(workers: Option<&WorkerSelection>) -> bool { let mode = tokenspeed_mm_tensor_transport_mode(); log_tokenspeed_transport_config_once(&mode); match mode.as_str() { // SHM only ever happens when SMG can actually write /dev/shm. "shm" => tokenspeed_shm_dev_writable(), - "auto" => worker_is_local(workers) && tokenspeed_shm_dev_writable(), + "auto" => worker_shares_dev_shm(workers) && tokenspeed_shm_dev_writable(), "" | "inline" => false, other => { log_unknown_tokenspeed_transport_once(other); @@ -1683,16 +1683,48 @@ fn log_unknown_tokenspeed_transport_once(value: &str) { }); } -/// A worker is treated as "local" (assumed to share SMG's `/dev/shm`) when its -/// gRPC URL targets a loopback host or a unix-domain socket. Conservative: an -/// unknown worker is non-local so `auto` falls back to the safe inline path. -fn worker_is_local(workers: Option<&WorkerSelection>) -> bool { +/// Whether the worker can be assumed to share SMG's `/dev/shm`, making the SHM +/// transport safe under `auto`. +/// +/// A unix-domain-socket worker is necessarily on the same host and, for the +/// sidecar/IPC topologies the SHM transport targets, shares `/dev/shm`, so it +/// qualifies. A TCP loopback worker only proves *network* locality — in common +/// sidecar/container setups the worker can be reachable on `127.0.0.1` while +/// using a private `/dev/shm`, and would then receive SHM handles it cannot +/// read — so loopback does NOT qualify unless the operator explicitly asserts a +/// shared namespace via `SMG_TOKENSPEED_SHM_ASSUME_LOOPBACK_SHARED`. Operators +/// that always co-locate can instead force the transport with the explicit +/// `shm` mode. An unknown worker is treated as non-sharing so `auto` falls back +/// to the safe inline path. +fn worker_shares_dev_shm(workers: Option<&WorkerSelection>) -> bool { let worker = match workers { Some(WorkerSelection::Single { worker }) => worker, Some(WorkerSelection::Dual { prefill, .. }) => prefill, None => return false, }; - url_is_loopback(worker.url()) + let url = worker.url(); + // A unix-domain socket proves same-host and (for co-located workers) a + // shared /dev/shm namespace. + if url.starts_with("unix:") { + return true; + } + // TCP loopback alone does not prove a shared /dev/shm; gate it behind an + // explicit operator assertion. + url_is_loopback(url) && tokenspeed_shm_assume_loopback_shared() +} + +/// Operator assertion that TCP-loopback TokenSpeed workers share SMG's +/// `/dev/shm` namespace. Off by default because loopback only proves network +/// locality, not a shared `/dev/shm`. +fn tokenspeed_shm_assume_loopback_shared() -> bool { + std::env::var("SMG_TOKENSPEED_SHM_ASSUME_LOOPBACK_SHARED") + .map(|value| { + matches!( + value.trim().to_ascii_lowercase().as_str(), + "1" | "true" | "yes" | "on" + ) + }) + .unwrap_or(false) } fn url_is_loopback(url: &str) -> bool { diff --git a/model_gateway/src/routers/grpc/proto_wrapper.rs b/model_gateway/src/routers/grpc/proto_wrapper.rs index 9bc3f1a6c1..e71bed26a3 100644 --- a/model_gateway/src/routers/grpc/proto_wrapper.rs +++ b/model_gateway/src/routers/grpc/proto_wrapper.rs @@ -668,11 +668,21 @@ fn collect_tokenspeed_tensor_shm_handles( } } +/// Prefix for every `/dev/shm` payload this transport creates +/// (see [`next_tokenspeed_shm_name`]). Cleanup only ever unlinks names +/// carrying this prefix so it cannot remove unrelated `/dev/shm` entries. +const TOKENSPEED_SHM_NAME_PREFIX: &str = "smg-tokenspeed-"; + fn validate_tokenspeed_shm_name_for_cleanup(name: &str) -> Option<&str> { let name = name.strip_prefix('/').unwrap_or(name); if name.is_empty() || name.contains('/') || name == "." || name == ".." || name.contains('\0') { return None; } + // Only unlink names this transport created; never touch arbitrary + // top-level /dev/shm entries. + if !name.starts_with(TOKENSPEED_SHM_NAME_PREFIX) { + return None; + } Some(name) } @@ -682,7 +692,13 @@ fn next_tokenspeed_shm_name() -> String { .duration_since(UNIX_EPOCH) .map(|duration| duration.as_nanos()) .unwrap_or_default(); - format!("smg-tokenspeed-{}-{}-{}", process::id(), nanos, seq) + format!( + "{}{}-{}-{}", + TOKENSPEED_SHM_NAME_PREFIX, + process::id(), + nanos, + seq + ) } fn tokenspeed_shm_path(name: &str) -> PathBuf { From 28fed55ae136e939849f617a67261f98a38f8fd1 Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 20:54:49 -0700 Subject: [PATCH 24/28] fix(grpc): hoist audio_url placeholder in String content format The String chat-content branch only matched image_url/video_url, so an audio_url part was silently dropped (no placeholder emitted) while the OpenAI branch already handled it. Match audio_url too, keeping the two content-format branches consistent and media-first. Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- model_gateway/src/routers/grpc/utils/chat_utils.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/model_gateway/src/routers/grpc/utils/chat_utils.rs b/model_gateway/src/routers/grpc/utils/chat_utils.rs index d327483bad..b0a0dfd1a5 100644 --- a/model_gateway/src/routers/grpc/utils/chat_utils.rs +++ b/model_gateway/src/routers/grpc/utils/chat_utils.rs @@ -245,7 +245,7 @@ fn transform_content_field( text_parts.push(t.to_string()); } } - Some("image_url") | Some("video_url") => { + Some("image_url") | Some("video_url") | Some("audio_url") => { if let Some(ph) = image_placeholder { media_parts.push(ph.to_string()); } From 278cab7d60b1f234b685445ded029e1ddd2741f9 Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 21:02:48 -0700 Subject: [PATCH 25/28] refactor(grpc): drop unsafe as_slice_memory_order fallback; document MM transport env MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - MM tensor serialization (serialize_array, write_array_as_f32/u16, serialize_array_as_u16_bytes) only fast-paths C-contiguous arrays now. The as_slice_memory_order() fallback serialized Fortran-contiguous views in the wrong dimension order — the exact hazard serialize_array's own comment warned about. Non-C-contiguous arrays fall through to logical .iter(); inputs are C-contiguous in practice, so the wire bytes are unchanged. - Refresh the stale 'worker is local' comment in assemble_tokenspeed to match worker_shares_dev_shm (unix socket / operator-asserted loopback). - Document the TokenSpeed MM tensor transport env vars (transport mode, shm min bytes, loopback assertion, timing) plus the worker-side companions in the configuration reference. Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- docs/reference/configuration.md | 18 +++++++++++ model_gateway/src/routers/grpc/multimodal.rs | 32 +++++++++++++++----- 2 files changed, 42 insertions(+), 8 deletions(-) diff --git a/docs/reference/configuration.md b/docs/reference/configuration.md index 7169a7e27c..76ca025267 100644 --- a/docs/reference/configuration.md +++ b/docs/reference/configuration.md @@ -936,3 +936,21 @@ smg \ | `JWT_AUDIENCE` | `--jwt-audience` | JWT audience | | `JWT_JWKS_URI` | `--jwt-jwks-uri` | JWKS URI | | `CONTROL_PLANE_API_KEYS` | `--control-plane-api-keys` | Control plane API keys | + +### TokenSpeed Multimodal Tensor Transport + +These env-only variables tune how the router ships preprocessed multimodal +tensors (image/video encoder inputs) to a TokenSpeed worker. They do not affect +accuracy — the inline and shared-memory paths produce byte-identical tensors. + +| Environment Variable | Default | Description | +|---------------------|---------|-------------| +| `SMG_TOKENSPEED_MM_TENSOR_TRANSPORT` | `inline` | Transport for large MM tensors: `inline` (gRPC bytes), `shm` (always use `/dev/shm`), or `auto` (use `/dev/shm` only when the worker is known to share it). Legacy alias: `SMG_TOKENSPEED_TENSOR_TRANSPORT`. | +| `SMG_TOKENSPEED_MM_SHM_MIN_BYTES` | `65536` | Minimum tensor size (bytes) before the SHM path is used; smaller tensors stay inline. Legacy alias: `SMG_TOKENSPEED_SHM_MIN_BYTES`. | +| `SMG_TOKENSPEED_SHM_ASSUME_LOOPBACK_SHARED` | `false` | In `auto` mode, treat TCP-loopback workers as sharing the router's `/dev/shm`. Off by default because loopback only proves network locality, not a shared namespace. Accepts `1`/`true`/`yes`/`on`. | +| `SMG_LOG_MM_TIMING` | `false` | Log per-stage multimodal preprocessing/assembly timing at `INFO`. Accepts `1`/`true`/`yes`. | + +The TokenSpeed gRPC servicer (worker side) reads two companion variables: +`TOKENSPEED_UNLINK_MM_SHM_AFTER_READ` (default on — unlink each `/dev/shm` +segment after the worker reads it) and `TOKENSPEED_LOG_MM_TIMING` (worker-side +timing logs). diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index 941fd2555e..c36f6e07d4 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -1063,7 +1063,8 @@ fn assemble_tokenspeed( let log_timing = log_mm_timing_enabled(); let total_started = Instant::now(); // Resolve the multimodal tensor transport once per request: `shm` always on, - // `auto` only when the worker is local (shares /dev/shm), otherwise inline. + // `auto` only when the worker is known to share /dev/shm (unix socket, or an + // operator-asserted loopback), otherwise inline. See `worker_shares_dev_shm`. let shm_enabled = resolve_tokenspeed_shm_enabled(workers); // Use patch-only offsets when available and non-empty; fall back to full structural ranges. let encoder_input_dtype = tokenspeed_encoder_input_dtype(intermediate.modality, workers); @@ -1344,8 +1345,12 @@ fn serialize_encoder_input(preprocessed: &PreprocessedEncoderInputs) -> (Vec fn serialize_array(encoder_input: &ArrayD) -> (Vec, Vec) { let encoder_bytes: Vec = if let Some(encoder_slice) = encoder_input + // Fast path only for C-contiguous arrays, whose memory order equals + // logical (row-major) order. A non-C-contiguous array (e.g. a + // Fortran-contiguous view) falls through to logical `.iter()` below; + // `as_slice_memory_order()` is deliberately NOT used as a fallback + // because it would serialize such arrays in the wrong dimension order. .as_slice() - .or_else(|| encoder_input.as_slice_memory_order()) { // Zero-copy reinterpret: &[f32] → &[u8] on little-endian (x86). // This replaces the per-element flat_map(to_le_bytes) which was the @@ -1360,9 +1365,8 @@ fn serialize_array(encoder_input: &ArrayD) -> (Vec, Vec) { encoder_slice.iter().flat_map(|v| v.to_le_bytes()).collect() } } else { - // Non-C-contiguous array: .iter() walks in logical (row-major) order, - // which matches the shape — unlike as_slice_memory_order() which would - // silently serialize in wrong dimension order for Fortran-contiguous arrays. + // Non-C-contiguous array: `.iter()` walks in logical (row-major) order, + // which matches the shape. encoder_input.iter().flat_map(|v| v.to_le_bytes()).collect() }; (encoder_bytes, array_shape(encoder_input)) @@ -1444,8 +1448,12 @@ fn write_array_as_dtype( fn write_array_as_f32(writer: &mut impl Write, encoder_input: &ArrayD) -> std::io::Result<()> { if let Some(encoder_slice) = encoder_input + // Fast path only for C-contiguous arrays, whose memory order equals + // logical (row-major) order. A non-C-contiguous array (e.g. a + // Fortran-contiguous view) falls through to logical `.iter()` below; + // `as_slice_memory_order()` is deliberately NOT used as a fallback + // because it would serialize such arrays in the wrong dimension order. .as_slice() - .or_else(|| encoder_input.as_slice_memory_order()) { return write_f32_slice(writer, encoder_slice); } @@ -1483,8 +1491,12 @@ where const CHUNK_VALUES: usize = 256 * 1024; if let Some(encoder_slice) = encoder_input + // Fast path only for C-contiguous arrays, whose memory order equals + // logical (row-major) order. A non-C-contiguous array (e.g. a + // Fortran-contiguous view) falls through to logical `.iter()` below; + // `as_slice_memory_order()` is deliberately NOT used as a fallback + // because it would serialize such arrays in the wrong dimension order. .as_slice() - .or_else(|| encoder_input.as_slice_memory_order()) { let mut converted: Vec = Vec::with_capacity(CHUNK_VALUES.min(encoder_slice.len())); for chunk in encoder_slice.chunks(CHUNK_VALUES) { @@ -1570,8 +1582,12 @@ where let mut converted = Vec::with_capacity(element_count); if let Some(encoder_slice) = encoder_input + // Fast path only for C-contiguous arrays, whose memory order equals + // logical (row-major) order. A non-C-contiguous array (e.g. a + // Fortran-contiguous view) falls through to logical `.iter()` below; + // `as_slice_memory_order()` is deliberately NOT used as a fallback + // because it would serialize such arrays in the wrong dimension order. .as_slice() - .or_else(|| encoder_input.as_slice_memory_order()) { converted.extend(encoder_slice.iter().map(|&value| convert(value))); } else { From 6d7a601b891b81fb14c67c588ac41459a5307e4e Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 21:07:15 -0700 Subject: [PATCH 26/28] refactor(grpc): drop self-introduced legacy MM transport env aliases MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit SMG_TOKENSPEED_TENSOR_TRANSPORT and SMG_TOKENSPEED_SHM_MIN_BYTES were introduced earlier in this same (unmerged) branch and then renamed to the SMG_TOKENSPEED_MM_* form for clarity, with the old names kept as 'legacy' fallbacks. Since neither name ever shipped on main, there is nothing to be backward-compatible with — drop the aliases and keep only the canonical SMG_TOKENSPEED_MM_* vars. Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- docs/reference/configuration.md | 4 ++-- model_gateway/src/routers/grpc/proto_wrapper.rs | 11 +++-------- 2 files changed, 5 insertions(+), 10 deletions(-) diff --git a/docs/reference/configuration.md b/docs/reference/configuration.md index 76ca025267..414ca4e685 100644 --- a/docs/reference/configuration.md +++ b/docs/reference/configuration.md @@ -945,8 +945,8 @@ accuracy — the inline and shared-memory paths produce byte-identical tensors. | Environment Variable | Default | Description | |---------------------|---------|-------------| -| `SMG_TOKENSPEED_MM_TENSOR_TRANSPORT` | `inline` | Transport for large MM tensors: `inline` (gRPC bytes), `shm` (always use `/dev/shm`), or `auto` (use `/dev/shm` only when the worker is known to share it). Legacy alias: `SMG_TOKENSPEED_TENSOR_TRANSPORT`. | -| `SMG_TOKENSPEED_MM_SHM_MIN_BYTES` | `65536` | Minimum tensor size (bytes) before the SHM path is used; smaller tensors stay inline. Legacy alias: `SMG_TOKENSPEED_SHM_MIN_BYTES`. | +| `SMG_TOKENSPEED_MM_TENSOR_TRANSPORT` | `inline` | Transport for large MM tensors: `inline` (gRPC bytes), `shm` (always use `/dev/shm`), or `auto` (use `/dev/shm` only when the worker is known to share it). | +| `SMG_TOKENSPEED_MM_SHM_MIN_BYTES` | `65536` | Minimum tensor size (bytes) before the SHM path is used; smaller tensors stay inline. | | `SMG_TOKENSPEED_SHM_ASSUME_LOOPBACK_SHARED` | `false` | In `auto` mode, treat TCP-loopback workers as sharing the router's `/dev/shm`. Off by default because loopback only proves network locality, not a shared namespace. Accepts `1`/`true`/`yes`/`on`. | | `SMG_LOG_MM_TIMING` | `false` | Log per-stage multimodal preprocessing/assembly timing at `INFO`. Accepts `1`/`true`/`yes`. | diff --git a/model_gateway/src/routers/grpc/proto_wrapper.rs b/model_gateway/src/routers/grpc/proto_wrapper.rs index e71bed26a3..b591aabbef 100644 --- a/model_gateway/src/routers/grpc/proto_wrapper.rs +++ b/model_gateway/src/routers/grpc/proto_wrapper.rs @@ -395,24 +395,19 @@ fn log_tokenspeed_mm_timing_enabled() -> bool { /// Multimodal tensor transport mode for the TokenSpeed backend. /// /// This only governs multimodal tensor payloads (encoder inputs and -/// model-specific tensors); prompt `input_ids` are always sent inline. The -/// canonical env var is `SMG_TOKENSPEED_MM_TENSOR_TRANSPORT`; the legacy -/// `SMG_TOKENSPEED_TENSOR_TRANSPORT` is still honored for backward -/// compatibility. +/// model-specific tensors); prompt `input_ids` are always sent inline. Set via +/// `SMG_TOKENSPEED_MM_TENSOR_TRANSPORT`. pub fn tokenspeed_mm_tensor_transport_mode() -> String { std::env::var("SMG_TOKENSPEED_MM_TENSOR_TRANSPORT") - .or_else(|_| std::env::var("SMG_TOKENSPEED_TENSOR_TRANSPORT")) .unwrap_or_default() .trim() .to_ascii_lowercase() } /// Minimum multimodal tensor size (bytes) before the SHM transport is used. -/// Canonical env var `SMG_TOKENSPEED_MM_SHM_MIN_BYTES`, with legacy fallback to -/// `SMG_TOKENSPEED_SHM_MIN_BYTES`. Defaults to 64 KiB. +/// Set via `SMG_TOKENSPEED_MM_SHM_MIN_BYTES`. Defaults to 64 KiB. pub fn tokenspeed_mm_shm_min_bytes() -> usize { std::env::var("SMG_TOKENSPEED_MM_SHM_MIN_BYTES") - .or_else(|_| std::env::var("SMG_TOKENSPEED_SHM_MIN_BYTES")) .ok() .and_then(|value| value.parse::().ok()) .unwrap_or(64 * 1024) From bc7da2886fab39c7fee46c9c81cb034cabb34dfd Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 21:31:45 -0700 Subject: [PATCH 27/28] feat(grpc): verify shared /dev/shm via filesystem-identity handshake MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit auto-mode SHM previously inferred a shared /dev/shm from the worker URL and fell back to an operator env (SMG_TOKENSPEED_SHM_ASSUME_LOOPBACK_SHARED). URL locality only proves network reachability — a loopback/sidecar worker can have a private /dev/shm — so the env pushed a guess onto operators. Instead negotiate it (NIXL-style): the TokenSpeed worker advertises its /dev/shm filesystem identity (:) in GetServerInfo's scheduler_info; discovery surfaces it as the worker's shm_namespace_id label; the router compares it to its own and enables SHM only on a match. boot_id pins the host; st_dev is the tmpfs superblock device, identical iff the same tmpfs backs both /dev/shm mounts — so it correctly detects sharing even across separate containers that share /dev/shm via --ipc/bind-mount (where mount namespaces differ but the superblock is the same). Any mismatch/missing token => inline. Verified on the live split-container deployment: a canary written to the router's /dev/shm is visible in the worker container, and both sides compute the identical token (:28) — i.e. token-match <=> actually-shared. (An earlier mount-namespace-inode token was rejected: it differs across those containers despite the shared tmpfs, which would wrongly force inline.) No new env, no proto change (reuses the existing scheduler_info Struct), and strictly more correct than URL inference. - client.rs: extract shm_namespace_id from scheduler_info into worker labels. - multimodal.rs: worker_shares_dev_shm compares the worker's token to the router's local :; remove url_is_loopback + the env; add a regression test that the local token resolves on Linux. - servicer.py: advertise _shm_namespace_id() in GetServerInfo. - docs: drop the loopback-assertion env; document the verified auto behavior. Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- docs/reference/configuration.md | 3 +- .../smg_grpc_servicer/tokenspeed/servicer.py | 24 ++++ model_gateway/src/routers/grpc/client.rs | 7 + model_gateway/src/routers/grpc/multimodal.rs | 124 +++++++++--------- 4 files changed, 95 insertions(+), 63 deletions(-) diff --git a/docs/reference/configuration.md b/docs/reference/configuration.md index 414ca4e685..67b9caef21 100644 --- a/docs/reference/configuration.md +++ b/docs/reference/configuration.md @@ -945,9 +945,8 @@ accuracy — the inline and shared-memory paths produce byte-identical tensors. | Environment Variable | Default | Description | |---------------------|---------|-------------| -| `SMG_TOKENSPEED_MM_TENSOR_TRANSPORT` | `inline` | Transport for large MM tensors: `inline` (gRPC bytes), `shm` (always use `/dev/shm`), or `auto` (use `/dev/shm` only when the worker is known to share it). | +| `SMG_TOKENSPEED_MM_TENSOR_TRANSPORT` | `inline` | Transport for large MM tensors: `inline` (gRPC bytes), `shm` (always use `/dev/shm`), or `auto` (use `/dev/shm` only when the worker is *verified* to share it). In `auto`, the router compares the worker's advertised `/dev/shm` namespace token (`GetServerInfo`) to its own and uses SHM only on a match; otherwise it falls back to inline. No locality configuration is needed. | | `SMG_TOKENSPEED_MM_SHM_MIN_BYTES` | `65536` | Minimum tensor size (bytes) before the SHM path is used; smaller tensors stay inline. | -| `SMG_TOKENSPEED_SHM_ASSUME_LOOPBACK_SHARED` | `false` | In `auto` mode, treat TCP-loopback workers as sharing the router's `/dev/shm`. Off by default because loopback only proves network locality, not a shared namespace. Accepts `1`/`true`/`yes`/`on`. | | `SMG_LOG_MM_TIMING` | `false` | Log per-stage multimodal preprocessing/assembly timing at `INFO`. Accepts `1`/`true`/`yes`. | The TokenSpeed gRPC servicer (worker side) reads two companion variables: diff --git a/grpc_servicer/smg_grpc_servicer/tokenspeed/servicer.py b/grpc_servicer/smg_grpc_servicer/tokenspeed/servicer.py index be2064c640..a2791c156d 100644 --- a/grpc_servicer/smg_grpc_servicer/tokenspeed/servicer.py +++ b/grpc_servicer/smg_grpc_servicer/tokenspeed/servicer.py @@ -468,6 +468,10 @@ async def GetServerInfo( scheduler_info_struct = Struct() scheduler_info_struct.update(_make_json_serializable(dict(self.scheduler_info))) + # Advertise this worker's /dev/shm namespace identity so the router can + # verify a shared /dev/shm before using the SHM tensor transport, instead + # of inferring locality from the worker URL. + scheduler_info_struct["shm_namespace_id"] = _shm_namespace_id() uptime = time.time() - self.start_time start_timestamp = Timestamp() @@ -1411,3 +1415,23 @@ def _make_json_serializable(obj: Any) -> Any: if isinstance(obj, dict): return {str(k): _make_json_serializable(v) for k, v in obj.items()} return str(obj) + + +def _shm_namespace_id() -> str: + """Identity of this process's ``/dev/shm`` tmpfs: ``:``. + + ``boot_id`` (``/proc/sys/kernel/random/boot_id``) is not namespaced, so it + pins the host; ``st_dev`` is the tmpfs superblock device backing + ``/dev/shm``. Two processes share ``/dev/shm`` iff both match — including + separate containers sharing it via ``--ipc``/bind-mount, where mount + namespaces differ but the underlying superblock (``st_dev``) is the same. The + router compares this token to its own to decide the SHM tensor transport. + Empty string if it can't be determined. + """ + try: + with open("/proc/sys/kernel/random/boot_id", encoding="ascii") as f: + boot_id = f.read().strip() + shm_dev = os.stat("/dev/shm").st_dev + return f"{boot_id}:{shm_dev}" + except OSError: + return "" diff --git a/model_gateway/src/routers/grpc/client.rs b/model_gateway/src/routers/grpc/client.rs index f9784ace80..0d83c03d28 100644 --- a/model_gateway/src/routers/grpc/client.rs +++ b/model_gateway/src/routers/grpc/client.rs @@ -806,6 +806,13 @@ impl ServerInfo { if !info.tokenspeed_version.is_empty() { labels.insert("version".to_string(), info.tokenspeed_version.clone()); } + // Carry the worker's /dev/shm namespace identity (advertised in + // scheduler_info). The router compares it to its own to decide the + // SHM tensor transport by *verifying* a shared /dev/shm rather than + // inferring it from the worker URL. See `worker_shares_dev_shm`. + if let Some(ref sched) = info.scheduler_info { + pick_prost_fields(&mut labels, sched, &["shm_namespace_id"]); + } labels } } diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index c36f6e07d4..b371c49b21 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -1063,8 +1063,8 @@ fn assemble_tokenspeed( let log_timing = log_mm_timing_enabled(); let total_started = Instant::now(); // Resolve the multimodal tensor transport once per request: `shm` always on, - // `auto` only when the worker is known to share /dev/shm (unix socket, or an - // operator-asserted loopback), otherwise inline. See `worker_shares_dev_shm`. + // `auto` only when the worker is verified to share /dev/shm (matching + // namespace token), otherwise inline. See `worker_shares_dev_shm`. let shm_enabled = resolve_tokenspeed_shm_enabled(workers); // Use patch-only offsets when available and non-empty; fall back to full structural ranges. let encoder_input_dtype = tokenspeed_encoder_input_dtype(intermediate.modality, workers); @@ -1699,75 +1699,59 @@ fn log_unknown_tokenspeed_transport_once(value: &str) { }); } -/// Whether the worker can be assumed to share SMG's `/dev/shm`, making the SHM +/// Whether the worker is *verified* to share SMG's `/dev/shm`, making the SHM /// transport safe under `auto`. /// -/// A unix-domain-socket worker is necessarily on the same host and, for the -/// sidecar/IPC topologies the SHM transport targets, shares `/dev/shm`, so it -/// qualifies. A TCP loopback worker only proves *network* locality — in common -/// sidecar/container setups the worker can be reachable on `127.0.0.1` while -/// using a private `/dev/shm`, and would then receive SHM handles it cannot -/// read — so loopback does NOT qualify unless the operator explicitly asserts a -/// shared namespace via `SMG_TOKENSPEED_SHM_ASSUME_LOOPBACK_SHARED`. Operators -/// that always co-locate can instead force the transport with the explicit -/// `shm` mode. An unknown worker is treated as non-sharing so `auto` falls back -/// to the safe inline path. +/// Rather than inferring locality from the worker URL (TCP loopback proves only +/// network locality, not a shared `/dev/shm`), the worker advertises its +/// `/dev/shm` filesystem identity (`:`) via +/// `GetServerInfo`, which discovery stores in the worker's `shm_namespace_id` +/// label. Two processes share `/dev/shm` iff these tokens match: `boot_id` pins +/// the host, and `st_dev` is the tmpfs superblock device, identical whenever the +/// same tmpfs backs both `/dev/shm` mounts — including separate containers that +/// share it via `--ipc`/bind-mount (where mount-namespace inodes differ but the +/// underlying superblock is the same). We compare the worker's token to ours: +/// equal ⇒ shared. A missing/empty token or any mismatch is treated as +/// non-sharing, so `auto` safely falls back to inline. fn worker_shares_dev_shm(workers: Option<&WorkerSelection>) -> bool { + let Some(local) = local_shm_namespace_id() else { + return false; + }; let worker = match workers { Some(WorkerSelection::Single { worker }) => worker, Some(WorkerSelection::Dual { prefill, .. }) => prefill, None => return false, }; - let url = worker.url(); - // A unix-domain socket proves same-host and (for co-located workers) a - // shared /dev/shm namespace. - if url.starts_with("unix:") { - return true; - } - // TCP loopback alone does not prove a shared /dev/shm; gate it behind an - // explicit operator assertion. - url_is_loopback(url) && tokenspeed_shm_assume_loopback_shared() -} - -/// Operator assertion that TCP-loopback TokenSpeed workers share SMG's -/// `/dev/shm` namespace. Off by default because loopback only proves network -/// locality, not a shared `/dev/shm`. -fn tokenspeed_shm_assume_loopback_shared() -> bool { - std::env::var("SMG_TOKENSPEED_SHM_ASSUME_LOOPBACK_SHARED") - .map(|value| { - matches!( - value.trim().to_ascii_lowercase().as_str(), - "1" | "true" | "yes" | "on" - ) - }) - .unwrap_or(false) + worker + .metadata() + .spec + .labels + .get("shm_namespace_id") + .is_some_and(|id| !id.is_empty() && id == local) } -fn url_is_loopback(url: &str) -> bool { - if url.starts_with("unix:") { - return true; - } - let rest = url.split_once("://").map(|(_, r)| r).unwrap_or(url); - let authority = rest.split('/').next().unwrap_or(rest); - let host = if let Some(stripped) = authority.strip_prefix('[') { - // IPv6 literal, e.g. [::1]:47165 - stripped.split(']').next().unwrap_or(stripped) - } else { - authority - .rsplit_once(':') - .map(|(h, _)| h) - .unwrap_or(authority) - } - .trim(); - host == "localhost" - || host - .parse::() - .map(|ip| ip.is_loopback()) - .unwrap_or(false) - || host - .parse::() - .map(|ip| ip.is_loopback()) - .unwrap_or(false) +/// This process's `/dev/shm` filesystem identity: `:`. +/// `boot_id` pins the host (it is not namespaced) and `st_dev` is the tmpfs +/// superblock device backing `/dev/shm`; together they identify the tmpfs so two +/// processes sharing it (even across containers via `--ipc`/bind-mount) produce +/// the same token. Computed once; `None` if it can't be determined (then `auto` +/// stays inline). +fn local_shm_namespace_id() -> Option<&'static str> { + static ID: OnceLock> = OnceLock::new(); + ID.get_or_init(compute_shm_namespace_id).as_deref() +} + +#[cfg(unix)] +fn compute_shm_namespace_id() -> Option { + use std::os::unix::fs::MetadataExt; + let boot_id = std::fs::read_to_string("/proc/sys/kernel/random/boot_id").ok()?; + let shm_dev = std::fs::metadata("/dev/shm").ok()?.dev(); + Some(format!("{}:{shm_dev}", boot_id.trim())) +} + +#[cfg(not(unix))] +fn compute_shm_namespace_id() -> Option { + None } fn canonical_float_dtype(dtype: &str) -> Option { @@ -1896,6 +1880,24 @@ mod tests { use super::*; + #[test] + #[cfg(unix)] + fn local_shm_namespace_id_resolves_on_linux() { + // /proc/.../boot_id and /dev/shm both exist on the Linux CI/runtime + // image, so the token must resolve to `:`. If it ever + // returned None, `auto` would silently never enable SHM. + let id = local_shm_namespace_id().expect("shm namespace id should resolve on Linux"); + assert!( + id.contains(':'), + "token must be :, got {id:?}" + ); + let dev = id.rsplit(':').next().unwrap(); + assert!( + dev.parse::().is_ok(), + "st_dev component must be numeric, got {id:?}" + ); + } + #[test] fn test_has_multimodal_content_with_images() { let messages = vec![ChatMessage::User { From 1c766de148dd2b55a752e77603ab99d7bff414bc Mon Sep 17 00:00:00 2001 From: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> Date: Sun, 21 Jun 2026 22:10:25 -0700 Subject: [PATCH 28/28] test(mm): guard PIL-exact video resize bit-identity MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The video resize path added for HF/vLLM parity had no test: the existing video tests use 4x4 frames that never resize, so the new resize_bicubic_pil_rgb / CatmullRom branch was never exercised — while images already have fingerprint + golden bit-identity guards. - transforms: resize_bicubic_pil_rgb must equal the DynamicImage resize_bicubic_pil byte-for-byte (the core invariant making video frames match images/HF/vLLM), plus a wrong-length-buffer rejection test. - qwen_vl_base: with a frame that actually needs resizing (odd 7x9 -> factor- aligned target), preprocess_video and preprocess_video_rgb must produce bit-identical encoder inputs; asserts a resize was forced so the branch runs. Signed-off-by: yechank-nvidia <161688079+yechank-nvidia@users.noreply.github.com> --- .../src/vision/processors/qwen_vl_base.rs | 79 +++++++++++++++++++ crates/multimodal/src/vision/transforms.rs | 42 ++++++++++ 2 files changed, 121 insertions(+) diff --git a/crates/multimodal/src/vision/processors/qwen_vl_base.rs b/crates/multimodal/src/vision/processors/qwen_vl_base.rs index c7c1c51243..18a7639abf 100644 --- a/crates/multimodal/src/vision/processors/qwen_vl_base.rs +++ b/crates/multimodal/src/vision/processors/qwen_vl_base.rs @@ -1079,6 +1079,85 @@ mod tests { DynamicImage::ImageRgb8(image) } + fn create_sized_pattern_frame(w: u32, h: u32, seed: u8) -> DynamicImage { + let mut image = RgbImage::new(w, h); + for y in 0..h { + for x in 0..w { + image.put_pixel( + x, + y, + image::Rgb([ + seed.wrapping_add((x * 3 + y) as u8), + seed.wrapping_add((x + y * 5) as u8), + seed.wrapping_add((x * 7 + y * 11) as u8), + ]), + ); + } + } + DynamicImage::ImageRgb8(image) + } + + /// When a video frame actually needs resizing, the DynamicImage path + /// (`preprocess_video` → `resize_bicubic_pil`) and the raw-RGB path + /// (`preprocess_video_rgb` → `resize_bicubic_pil_rgb`) must produce + /// byte-for-byte identical encoder inputs. The other video tests use 4x4 + /// frames that need no resize, so this is the only one exercising the + /// default-bicubic resize branch added for HF/vLLM parity. + #[test] + fn test_preprocess_video_rgb_matches_dynamic_with_resize() { + let processor = QwenVLProcessorBase::new(create_video_test_config()); + let config = PreProcessorConfig { + image_mean: Some(processor.default_mean().to_vec()), + image_std: Some(processor.default_std().to_vec()), + ..Default::default() + }; + // Odd dimensions force smart_resize_video to a different factor-aligned + // target, guaranteeing the resize branch runs. + let frames = vec![ + create_sized_pattern_frame(7, 9, 3), + create_sized_pattern_frame(7, 9, 101), + ]; + let (target_h, target_w) = processor.smart_resize_video(frames.len(), 9, 7).unwrap(); + assert!( + (target_w as u32, target_h as u32) != (7u32, 9u32), + "test must force a resize; target {target_w}x{target_h} should differ from 7x9" + ); + + let rgb_frames = frames + .iter() + .map(|frame| { + let DynamicImage::ImageRgb8(rgb) = frame else { + panic!("test frame is not RGB8"); + }; + RgbFrameRef { + width: rgb.width(), + height: rgb.height(), + data: rgb.as_raw(), + } + }) + .collect::>(); + + let dynamic = processor.preprocess_video(&frames, &config).unwrap(); + let rgb = processor + .preprocess_video_rgb(&rgb_frames, &config) + .unwrap(); + + let a = dynamic.encoder_input.as_slice_memory_order().unwrap(); + let b = rgb.encoder_input.as_slice_memory_order().unwrap(); + assert_eq!( + a.len(), + b.len(), + "resized video encoder input length differs" + ); + for (idx, (&got, &want)) in a.iter().zip(b.iter()).enumerate() { + assert_eq!( + got.to_bits(), + want.to_bits(), + "resized video path diverges at index {idx}: dynamic {got} vs rgb {want}" + ); + } + } + #[test] fn test_qwen_vl_base_factor() { let processor = QwenVLProcessorBase::new(create_test_config()); diff --git a/crates/multimodal/src/vision/transforms.rs b/crates/multimodal/src/vision/transforms.rs index f8063fc7c7..06a8e091f6 100644 --- a/crates/multimodal/src/vision/transforms.rs +++ b/crates/multimodal/src/vision/transforms.rs @@ -875,6 +875,48 @@ mod tests { DynamicImage::from(RgbImage::from_pixel(width, height, color)) } + /// The raw-RGB video resizer must be byte-for-byte identical to the + /// DynamicImage PIL-bicubic resizer used for images, so default-bicubic + /// video frames match HF/vLLM exactly — the same bit-identity guarantee + /// images get via the fingerprint tests. Guards the video resize path added + /// for HF/vLLM parity (`preprocess_video_rgb`). + #[test] + fn resize_bicubic_pil_rgb_matches_dynamic_path() { + let (src_w, src_h) = (37u32, 23u32); // non-aligned source, non-trivial ratios + let (out_w, out_h) = (16u32, 28u32); // downscale width, upscale height + let mut img = RgbImage::new(src_w, src_h); + for y in 0..src_h { + for x in 0..src_w { + img.put_pixel( + x, + y, + Rgb([ + ((x * 7) ^ (y * 13)) as u8, + (x * 3 + y * 5) as u8, + (x + y * y) as u8, + ]), + ); + } + } + let via_dynamic = resize_bicubic_pil(&DynamicImage::ImageRgb8(img.clone()), out_w, out_h); + let via_bytes = resize_bicubic_pil_rgb(img.as_raw(), src_w, src_h, out_w, out_h).unwrap(); + assert_eq!( + via_dynamic.to_rgb8().into_raw(), + via_bytes.into_raw(), + "raw-RGB PIL bicubic must equal DynamicImage PIL bicubic byte-for-byte" + ); + } + + /// `resize_bicubic_pil_rgb` rejects a buffer whose length doesn't match the + /// declared dimensions rather than reading out of bounds. + #[test] + fn resize_bicubic_pil_rgb_rejects_wrong_length() { + assert!( + resize_bicubic_pil_rgb(&[0u8; 10], 4, 4, 2, 2).is_err(), + "wrong-length RGB buffer must error, not panic" + ); + } + #[test] fn test_to_tensor_shape() { let img = create_test_image(10, 20, Rgb([255, 128, 0]));