From 6a824e7fc7db3217a801f3bdbbb552171bab2d9e Mon Sep 17 00:00:00 2001 From: Chang Su Date: Wed, 1 Apr 2026 08:01:40 +0000 Subject: [PATCH 1/5] perf(multimodal): add profiling profile and preprocessing timing instrumentation Signed-off-by: Chang Su --- Cargo.toml | 5 ++++ crates/multimodal/Cargo.toml | 1 + .../src/vision/processors/llama4_vision.rs | 27 ++++++++++++++++--- model_gateway/src/routers/grpc/multimodal.rs | 6 +++++ 4 files changed, 35 insertions(+), 4 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index 7645f8ddde..b9d3e1f8f1 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -126,6 +126,11 @@ lto = "fat" # Full LTO for smaller binaries codegen-units = 1 # Better optimization, slower compile strip = true # Strip debug symbols +[profile.profiling] +inherits = "release" +debug = 1 +strip = false + [profile.ci] inherits = "release" opt-level = 2 # Lighter optimization (still fast runtime, much faster compile) diff --git a/crates/multimodal/Cargo.toml b/crates/multimodal/Cargo.toml index 6cfea0509f..962f9db30f 100644 --- a/crates/multimodal/Cargo.toml +++ b/crates/multimodal/Cargo.toml @@ -35,6 +35,7 @@ url = "2.5.4" llm-tokenizer.workspace = true anyhow.workspace = true blake3.workspace = true +tracing.workspace = true [dev-dependencies] criterion = { version = "0.8", features = ["html_reports"] } diff --git a/crates/multimodal/src/vision/processors/llama4_vision.rs b/crates/multimodal/src/vision/processors/llama4_vision.rs index 7cfa60a8a4..1d4678a90c 100644 --- a/crates/multimodal/src/vision/processors/llama4_vision.rs +++ b/crates/multimodal/src/vision/processors/llama4_vision.rs @@ -326,7 +326,6 @@ impl Llama4VisionProcessor { let (target_h, target_w) = target_size; // Step 2: Compute resize target - limit upscaling if not resize_to_max_canvas - // This limits how much we resize the image, but we still pad to target_size let resize_target = if self.resize_to_max_canvas { target_size } else { @@ -340,24 +339,30 @@ impl Llama4VisionProcessor { let new_size = Self::get_max_res_without_distortion(image_size, resize_target); let (new_h, new_w) = (new_size.0.max(1), new_size.1.max(1)); + let t_resize = std::time::Instant::now(); let resized = transforms::resize(image, new_w, new_h, FilterType::Triangle); + let resize_us = t_resize.elapsed().as_micros(); - // Step 4: Pad to target_size (the canvas from get_best_fit, not resize_target) + // Step 4: Pad to target_size + let t_pad = std::time::Instant::now(); let padded = self.pad_image(&resized, target_w, target_h); + let pad_us = t_pad.elapsed().as_micros(); // Step 5: Convert to tensor and normalize + let t_tensor = std::time::Instant::now(); let tensor = transforms::to_tensor_and_normalize(&padded, &self.mean, &self.std); + let tensor_us = t_tensor.elapsed().as_micros(); // Step 6: Calculate tile counts based on target_size (canvas size) let tile = self.tile_size as usize; let num_tiles_h = target_h as usize / tile; let num_tiles_w = target_w as usize / tile; - // Step 7: Split into tiles + // Step 7: Split into tiles + global tile + let t_tile = std::time::Instant::now(); let tiles = self.split_to_tiles(&tensor, num_tiles_h, num_tiles_w); let num_tiles = num_tiles_h * num_tiles_w; - // Step 8: Add global tile if there are multiple tiles let output = if num_tiles > 1 { let global_tile = self.create_global_image(image); let mut combined = Array4::::zeros((num_tiles + 1, 3, tile, tile)); @@ -371,6 +376,20 @@ impl Llama4VisionProcessor { } else { tiles }; + let tile_us = t_tile.elapsed().as_micros(); + + let total_us = resize_us + pad_us + tensor_us + tile_us; + if total_us > 5000 { + tracing::debug!( + model = "llama4-vision", + resize_us = %resize_us, + pad_us = %pad_us, + tensor_us = %tensor_us, + tile_us = %tile_us, + total_us = %total_us, + "preprocess timing breakdown" + ); + } (output, (num_tiles_h, num_tiles_w)) } diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index aad81fe4f1..4a5b1734f4 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -413,6 +413,7 @@ async fn process_multimodal_parts( ); // Step 2: Resolve model spec and preprocess images + let t_config = std::time::Instant::now(); let model_config = components .get_or_load_config(model_id, tokenizer_source) .await?; @@ -434,18 +435,23 @@ async fn process_multimodal_parts( .image_processor_registry .find(model_id, model_type) .ok_or_else(|| anyhow::anyhow!("No image processor found for model: {model_id}"))?; + let config_us = t_config.elapsed().as_micros(); // ImagePreProcessor::preprocess takes &[DynamicImage]; images are behind Arc. // Clone cost is negligible vs. the preprocessing work itself. + let t_preprocess = std::time::Instant::now(); let dynamic_images: Vec = images.iter().map(|f| f.image.clone()).collect(); let preprocessed: PreprocessedImages = image_processor .preprocess(&dynamic_images, &model_config.preprocessor_config) .map_err(|e| anyhow::anyhow!("Image preprocessing failed: {e}"))?; + let preprocess_us = t_preprocess.elapsed().as_micros(); debug!( num_images = preprocessed.num_img_tokens.len(), total_tokens = preprocessed.num_img_tokens.iter().sum::(), + config_us, + preprocess_us, "Image preprocessing complete" ); From 1c91aca034d85e686f7950c85de9a56db1f7a0a1 Mon Sep 17 00:00:00 2001 From: Chang Su Date: Wed, 1 Apr 2026 15:03:34 +0000 Subject: [PATCH 2/5] perf(multimodal): optimize serialize_pixel_values, build_planar_tensor, and pad+tensor fusion Three optimizations based on perf profiling of MMMU benchmark: 1. serialize_pixel_values: replace per-element flat_map(to_le_bytes) with bytemuck::cast_slice zero-copy reinterpret. Eliminates FlattenCompat which was #1 CPU hotspot (24-31% of SMG CPU). 2. build_planar_tensor: restructure RGB deinterleave loop into 8-pixel blocks for better auto-vectorization. Reduces tensor conversion from 22% to 5% of CPU. 3. Llama4 pad_and_normalize_to_tensor: fuse pad_image + to_tensor into one pass, eliminating intermediate padded RgbImage allocation. Removes 13% CPU overhead from image::overlay + get_pixel. Combined effect on Llama4 preprocessing: avg 25.5ms -> 16.1ms (-37%), median 21.6ms -> 9.8ms (-55%). MMMU accuracy unchanged (0.417 -> 0.428). Signed-off-by: Chang Su --- Cargo.toml | 4 +- .../src/vision/processors/llama4_vision.rs | 67 ++++---------- crates/multimodal/src/vision/transforms.rs | 91 ++++++++++++++++++- model_gateway/src/routers/grpc/multimodal.rs | 18 +++- 4 files changed, 122 insertions(+), 58 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index b9d3e1f8f1..5a255b89dd 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -128,8 +128,8 @@ strip = true # Strip debug symbols [profile.profiling] inherits = "release" -debug = 1 -strip = false +debug = 1 # Line-level debug info for flamegraphs +strip = false # Keep symbols [profile.ci] inherits = "release" diff --git a/crates/multimodal/src/vision/processors/llama4_vision.rs b/crates/multimodal/src/vision/processors/llama4_vision.rs index 1d4678a90c..32e0a935d0 100644 --- a/crates/multimodal/src/vision/processors/llama4_vision.rs +++ b/crates/multimodal/src/vision/processors/llama4_vision.rs @@ -30,7 +30,7 @@ use std::collections::HashSet; -use image::{imageops::FilterType, DynamicImage, GenericImageView, Rgb, RgbImage}; +use image::{imageops::FilterType, DynamicImage, GenericImageView}; use ndarray::{s, Array3, Array4}; use crate::vision::{ @@ -258,30 +258,6 @@ impl Llama4VisionProcessor { } } - /// Pad image to target dimensions with black padding. - #[expect( - clippy::unused_self, - reason = "method logically belongs to the processor; keeps API consistent" - )] - fn pad_image(&self, image: &DynamicImage, target_w: u32, target_h: u32) -> DynamicImage { - let (w, h) = image.dimensions(); - if w == target_w && h == target_h { - return image.clone(); - } - - // Create black background (LLaMA 4 uses 0 for padding) - let black = Rgb([0u8, 0, 0]); - let mut padded = RgbImage::from_pixel(target_w, target_h, black); - - // Copy image to top-left — avoid to_rgb8() if already RGB8 - match image { - DynamicImage::ImageRgb8(rgb) => image::imageops::overlay(&mut padded, rgb, 0, 0), - _ => image::imageops::overlay(&mut padded, &image.to_rgb8(), 0, 0), - } - - DynamicImage::ImageRgb8(padded) - } - /// Split image tensor into tiles. fn split_to_tiles( &self, @@ -339,19 +315,21 @@ impl Llama4VisionProcessor { let new_size = Self::get_max_res_without_distortion(image_size, resize_target); let (new_h, new_w) = (new_size.0.max(1), new_size.1.max(1)); - let t_resize = std::time::Instant::now(); let resized = transforms::resize(image, new_w, new_h, FilterType::Triangle); - let resize_us = t_resize.elapsed().as_micros(); - // Step 4: Pad to target_size - let t_pad = std::time::Instant::now(); - let padded = self.pad_image(&resized, target_w, target_h); - let pad_us = t_pad.elapsed().as_micros(); - - // Step 5: Convert to tensor and normalize - let t_tensor = std::time::Instant::now(); - let tensor = transforms::to_tensor_and_normalize(&padded, &self.mean, &self.std); - let tensor_us = t_tensor.elapsed().as_micros(); + // Fused pad + tensor: build the padded f32 tensor directly from the + // resized RGB bytes, avoiding an intermediate padded RgbImage allocation. + let tensor = if new_w != target_w || new_h != target_h { + transforms::pad_and_normalize_to_tensor( + &resized, + target_w as usize, + target_h as usize, + &self.mean, + &self.std, + ) + } else { + transforms::to_tensor_and_normalize(&resized, &self.mean, &self.std) + }; // Step 6: Calculate tile counts based on target_size (canvas size) let tile = self.tile_size as usize; @@ -359,7 +337,6 @@ impl Llama4VisionProcessor { let num_tiles_w = target_w as usize / tile; // Step 7: Split into tiles + global tile - let t_tile = std::time::Instant::now(); let tiles = self.split_to_tiles(&tensor, num_tiles_h, num_tiles_w); let num_tiles = num_tiles_h * num_tiles_w; @@ -376,20 +353,6 @@ impl Llama4VisionProcessor { } else { tiles }; - let tile_us = t_tile.elapsed().as_micros(); - - let total_us = resize_us + pad_us + tensor_us + tile_us; - if total_us > 5000 { - tracing::debug!( - model = "llama4-vision", - resize_us = %resize_us, - pad_us = %pad_us, - tensor_us = %tensor_us, - tile_us = %tile_us, - total_us = %total_us, - "preprocess timing breakdown" - ); - } (output, (num_tiles_h, num_tiles_w)) } @@ -527,6 +490,8 @@ impl ImagePreProcessor for Llama4VisionProcessor { #[cfg(test)] mod tests { + use image::{Rgb, RgbImage}; + use super::*; fn create_test_image(width: u32, height: u32, color: Rgb) -> DynamicImage { diff --git a/crates/multimodal/src/vision/transforms.rs b/crates/multimodal/src/vision/transforms.rs index d9fdff7c02..c282b824c1 100644 --- a/crates/multimodal/src/vision/transforms.rs +++ b/crates/multimodal/src/vision/transforms.rs @@ -68,10 +68,35 @@ fn build_planar_tensor( let (r_plane, rest) = data.split_at_mut(pixels); let (g_plane, b_plane) = rest.split_at_mut(pixels); - for (i, chunk) in raw.chunks_exact(3).enumerate() { - r_plane[i] = chunk[0] as f32 * scale[0] + bias[0]; - g_plane[i] = chunk[1] as f32 * scale[1] + bias[1]; - b_plane[i] = chunk[2] as f32 * scale[2] + bias[2]; + // Deinterleave RGB → planar f32 with scale+bias. + // Process 8 pixels (24 bytes) at a time for better auto-vectorization. + // The fixed inner loop lets the compiler unroll and use SIMD gather+convert. + let full_blocks = pixels / 8; + let remainder = pixels % 8; + + for block in 0..full_blocks { + let dst = block * 8; + let src_base = dst * 3; + let src = &raw[src_base..src_base + 24]; + let rd = &mut r_plane[dst..dst + 8]; + let gd = &mut g_plane[dst..dst + 8]; + let bd = &mut b_plane[dst..dst + 8]; + + for i in 0..8 { + let s = i * 3; + rd[i] = src[s] as f32 * scale[0] + bias[0]; + gd[i] = src[s + 1] as f32 * scale[1] + bias[1]; + bd[i] = src[s + 2] as f32 * scale[2] + bias[2]; + } + } + + let tail_dst = full_blocks * 8; + let tail_src = tail_dst * 3; + for i in 0..remainder { + let s = tail_src + i * 3; + r_plane[tail_dst + i] = raw[s] as f32 * scale[0] + bias[0]; + g_plane[tail_dst + i] = raw[s + 1] as f32 * scale[1] + bias[1]; + b_plane[tail_dst + i] = raw[s + 2] as f32 * scale[2] + bias[2]; } #[expect( @@ -342,6 +367,64 @@ pub fn pil_to_filter(resampling: Option) -> FilterType { } } +/// Build a padded [C, H, W] f32 tensor from a smaller image. +/// +/// The image is placed at top-left, and the remaining canvas is filled with +/// the normalized value of black (0). This fuses `pad_image` + `to_tensor_and_normalize` +/// into one step, avoiding an intermediate padded `RgbImage` allocation. +/// +/// This is used by LLaMA 4 which pads with black (0) before tiling. +pub fn pad_and_normalize_to_tensor( + image: &DynamicImage, + canvas_w: usize, + canvas_h: usize, + mean: &[f64; 3], + std: &[f64; 3], +) -> Array3 { + let (img_w, img_h, raw) = rgb_bytes(image); + let canvas_pixels = canvas_h * canvas_w; + + // Precompute fused scale/bias: (pixel/255 - mean) / std + let scale: [f32; 3] = std::array::from_fn(|c| 1.0 / (255.0 * std[c] as f32)); + let bias: [f32; 3] = std::array::from_fn(|c| -(mean[c] as f32) / (std[c] as f32)); + // Normalized value of black (0): 0 * scale + bias = bias + let pad_val = bias; + + let mut data = vec![0.0f32; 3 * canvas_pixels]; + let (r_plane, rest) = data.split_at_mut(canvas_pixels); + let (g_plane, b_plane) = rest.split_at_mut(canvas_pixels); + + // Pre-fill with padding value + r_plane.fill(pad_val[0]); + g_plane.fill(pad_val[1]); + b_plane.fill(pad_val[2]); + + // Overwrite image region row-by-row + let rw = img_w.min(canvas_w); + let rh = img_h.min(canvas_h); + for y in 0..rh { + let src_row = &raw[y * img_w * 3..y * img_w * 3 + rw * 3]; + let dst_offset = y * canvas_w; + let rd = &mut r_plane[dst_offset..dst_offset + rw]; + let gd = &mut g_plane[dst_offset..dst_offset + rw]; + let bd = &mut b_plane[dst_offset..dst_offset + rw]; + + for x in 0..rw { + let s = x * 3; + rd[x] = src_row[s] as f32 * scale[0] + bias[0]; + gd[x] = src_row[s + 1] as f32 * scale[1] + bias[1]; + bd[x] = src_row[s + 2] as f32 * scale[2] + bias[2]; + } + } + + #[expect( + clippy::expect_used, + reason = "data has exactly 3*canvas_h*canvas_w elements by construction" + )] + Array3::from_shape_vec((3, canvas_h, canvas_w), data) + .expect("shape matches pre-allocated buffer") +} + /// Calculate mean color of an image as RGB. pub fn calculate_mean_color(image: &DynamicImage) -> Rgb { let rgb = image.to_rgb8(); diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index 4a5b1734f4..70a8fa998e 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -407,8 +407,13 @@ async fn process_multimodal_parts( )); } + let image_sizes: Vec<(u32, u32)> = images + .iter() + .map(|f| (f.image.width(), f.image.height())) + .collect(); debug!( image_count = images.len(), + ?image_sizes, "Fetched images for multimodal processing" ); @@ -690,7 +695,18 @@ fn serialize_pixel_values(preprocessed: &PreprocessedImages) -> (Vec, Vec Date: Wed, 1 Apr 2026 15:22:47 +0000 Subject: [PATCH 3/5] chore: remove profiling profile and unused tracing dep from multimodal Signed-off-by: Chang Su --- Cargo.toml | 5 ----- crates/multimodal/Cargo.toml | 1 - 2 files changed, 6 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index 5a255b89dd..7645f8ddde 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -126,11 +126,6 @@ lto = "fat" # Full LTO for smaller binaries codegen-units = 1 # Better optimization, slower compile strip = true # Strip debug symbols -[profile.profiling] -inherits = "release" -debug = 1 # Line-level debug info for flamegraphs -strip = false # Keep symbols - [profile.ci] inherits = "release" opt-level = 2 # Lighter optimization (still fast runtime, much faster compile) diff --git a/crates/multimodal/Cargo.toml b/crates/multimodal/Cargo.toml index 962f9db30f..6cfea0509f 100644 --- a/crates/multimodal/Cargo.toml +++ b/crates/multimodal/Cargo.toml @@ -35,7 +35,6 @@ url = "2.5.4" llm-tokenizer.workspace = true anyhow.workspace = true blake3.workspace = true -tracing.workspace = true [dev-dependencies] criterion = { version = "0.8", features = ["html_reports"] } From 56f77279d550cc48d2d5dc5a22215e05dd479263 Mon Sep 17 00:00:00 2001 From: Chang Su Date: Wed, 1 Apr 2026 15:31:53 +0000 Subject: [PATCH 4/5] refactor(multimodal): extract deinterleave helper, move pad fusion to llama4, remove timing - Extract `deinterleave_rgb_to_planes` as shared public helper with 8-pixel block optimization, used by both `build_planar_tensor` and Llama4's `pad_and_normalize_to_tensor` - Move `pad_and_normalize_to_tensor` from transforms.rs to llama4_vision.rs since it's only used there - Remove eager `image_sizes` allocation from hot path (review feedback) - Remove timing instrumentation from gateway multimodal handler Signed-off-by: Chang Su --- .../src/vision/processors/llama4_vision.rs | 59 +++++++-- crates/multimodal/src/vision/transforms.rs | 114 ++++++------------ model_gateway/src/routers/grpc/multimodal.rs | 11 -- 3 files changed, 89 insertions(+), 95 deletions(-) diff --git a/crates/multimodal/src/vision/processors/llama4_vision.rs b/crates/multimodal/src/vision/processors/llama4_vision.rs index 32e0a935d0..df1362ac9e 100644 --- a/crates/multimodal/src/vision/processors/llama4_vision.rs +++ b/crates/multimodal/src/vision/processors/llama4_vision.rs @@ -258,6 +258,57 @@ impl Llama4VisionProcessor { } } + /// Build a padded [C, H, W] f32 tensor from a smaller image. + /// + /// The image is placed at top-left, and the remaining canvas is filled with + /// the normalized value of black (0). This fuses pad + tensor conversion + /// into one step, avoiding an intermediate padded `RgbImage` allocation. + fn pad_and_normalize_to_tensor( + &self, + image: &DynamicImage, + canvas_w: usize, + canvas_h: usize, + ) -> Array3 { + let (img_w, img_h, raw) = transforms::rgb_bytes(image); + let canvas_pixels = canvas_h * canvas_w; + + // Precompute fused scale/bias: (pixel/255 - mean) / std + let scale: [f32; 3] = std::array::from_fn(|c| 1.0 / (255.0 * self.std[c] as f32)); + let bias: [f32; 3] = std::array::from_fn(|c| -(self.mean[c] as f32) / (self.std[c] as f32)); + + let mut data = vec![0.0f32; 3 * canvas_pixels]; + let (r_plane, rest) = data.split_at_mut(canvas_pixels); + let (g_plane, b_plane) = rest.split_at_mut(canvas_pixels); + + // Pre-fill with normalized black: 0 * scale + bias = bias + r_plane.fill(bias[0]); + g_plane.fill(bias[1]); + b_plane.fill(bias[2]); + + // Overwrite image region row-by-row using the shared block-optimized helper + let rw = img_w.min(canvas_w); + let rh = img_h.min(canvas_h); + for y in 0..rh { + let src_row = &raw[y * img_w * 3..y * img_w * 3 + rw * 3]; + let dst_offset = y * canvas_w; + transforms::deinterleave_rgb_to_planes( + src_row, + &mut r_plane[dst_offset..dst_offset + rw], + &mut g_plane[dst_offset..dst_offset + rw], + &mut b_plane[dst_offset..dst_offset + rw], + scale, + bias, + ); + } + + #[expect( + clippy::expect_used, + reason = "data has exactly 3*canvas_h*canvas_w elements by construction" + )] + Array3::from_shape_vec((3, canvas_h, canvas_w), data) + .expect("shape matches pre-allocated buffer") + } + /// Split image tensor into tiles. fn split_to_tiles( &self, @@ -320,13 +371,7 @@ impl Llama4VisionProcessor { // Fused pad + tensor: build the padded f32 tensor directly from the // resized RGB bytes, avoiding an intermediate padded RgbImage allocation. let tensor = if new_w != target_w || new_h != target_h { - transforms::pad_and_normalize_to_tensor( - &resized, - target_w as usize, - target_h as usize, - &self.mean, - &self.std, - ) + self.pad_and_normalize_to_tensor(&resized, target_w as usize, target_h as usize) } else { transforms::to_tensor_and_normalize(&resized, &self.mean, &self.std) }; diff --git a/crates/multimodal/src/vision/transforms.rs b/crates/multimodal/src/vision/transforms.rs index c282b824c1..c4b53d32cc 100644 --- a/crates/multimodal/src/vision/transforms.rs +++ b/crates/multimodal/src/vision/transforms.rs @@ -38,7 +38,7 @@ pub type Result = std::result::Result; /// Extract RGB pixel data from a DynamicImage, avoiding a copy when already RGB8. /// Returns (width, height, raw_bytes) where raw_bytes is interleaved R,G,B,R,G,B,... -fn rgb_bytes(image: &DynamicImage) -> (usize, usize, std::borrow::Cow<'_, [u8]>) { +pub fn rgb_bytes(image: &DynamicImage) -> (usize, usize, std::borrow::Cow<'_, [u8]>) { match image { DynamicImage::ImageRgb8(rgb) => ( rgb.width() as usize, @@ -54,30 +54,31 @@ fn rgb_bytes(image: &DynamicImage) -> (usize, usize, std::borrow::Cow<'_, [u8]>) } } -/// Build a [C, H, W] f32 tensor from interleaved RGB bytes with per-channel -/// `scale` and `bias`: `output[c][i] = raw[i*3 + c] * scale[c] + bias[c]`. -fn build_planar_tensor( - raw: &[u8], - w: usize, - h: usize, +/// Deinterleave interleaved RGB bytes into separate R, G, B f32 planes with +/// per-channel `scale` and `bias`: `plane[c][i] = rgb[i*3 + c] * scale[c] + bias[c]`. +/// +/// Processes 8 pixels at a time so the compiler can unroll and auto-vectorize +/// the stride-3 gather pattern. +pub fn deinterleave_rgb_to_planes( + rgb: &[u8], + r_plane: &mut [f32], + g_plane: &mut [f32], + b_plane: &mut [f32], scale: [f32; 3], bias: [f32; 3], -) -> Array3 { - let pixels = h * w; - let mut data = vec![0.0f32; 3 * pixels]; - let (r_plane, rest) = data.split_at_mut(pixels); - let (g_plane, b_plane) = rest.split_at_mut(pixels); +) { + let pixels = r_plane.len(); + debug_assert_eq!(pixels, g_plane.len()); + debug_assert_eq!(pixels, b_plane.len()); + debug_assert!(rgb.len() >= pixels * 3); - // Deinterleave RGB → planar f32 with scale+bias. - // Process 8 pixels (24 bytes) at a time for better auto-vectorization. - // The fixed inner loop lets the compiler unroll and use SIMD gather+convert. let full_blocks = pixels / 8; let remainder = pixels % 8; for block in 0..full_blocks { let dst = block * 8; let src_base = dst * 3; - let src = &raw[src_base..src_base + 24]; + let src = &rgb[src_base..src_base + 24]; let rd = &mut r_plane[dst..dst + 8]; let gd = &mut g_plane[dst..dst + 8]; let bd = &mut b_plane[dst..dst + 8]; @@ -94,10 +95,27 @@ fn build_planar_tensor( let tail_src = tail_dst * 3; for i in 0..remainder { let s = tail_src + i * 3; - r_plane[tail_dst + i] = raw[s] as f32 * scale[0] + bias[0]; - g_plane[tail_dst + i] = raw[s + 1] as f32 * scale[1] + bias[1]; - b_plane[tail_dst + i] = raw[s + 2] as f32 * scale[2] + bias[2]; + r_plane[tail_dst + i] = rgb[s] as f32 * scale[0] + bias[0]; + g_plane[tail_dst + i] = rgb[s + 1] as f32 * scale[1] + bias[1]; + b_plane[tail_dst + i] = rgb[s + 2] as f32 * scale[2] + bias[2]; } +} + +/// Build a [C, H, W] f32 tensor from interleaved RGB bytes with per-channel +/// `scale` and `bias`: `output[c][i] = raw[i*3 + c] * scale[c] + bias[c]`. +fn build_planar_tensor( + raw: &[u8], + w: usize, + h: usize, + scale: [f32; 3], + bias: [f32; 3], +) -> Array3 { + let pixels = h * w; + let mut data = vec![0.0f32; 3 * pixels]; + let (r_plane, rest) = data.split_at_mut(pixels); + let (g_plane, b_plane) = rest.split_at_mut(pixels); + + deinterleave_rgb_to_planes(raw, r_plane, g_plane, b_plane, scale, bias); #[expect( clippy::expect_used, @@ -367,64 +385,6 @@ pub fn pil_to_filter(resampling: Option) -> FilterType { } } -/// Build a padded [C, H, W] f32 tensor from a smaller image. -/// -/// The image is placed at top-left, and the remaining canvas is filled with -/// the normalized value of black (0). This fuses `pad_image` + `to_tensor_and_normalize` -/// into one step, avoiding an intermediate padded `RgbImage` allocation. -/// -/// This is used by LLaMA 4 which pads with black (0) before tiling. -pub fn pad_and_normalize_to_tensor( - image: &DynamicImage, - canvas_w: usize, - canvas_h: usize, - mean: &[f64; 3], - std: &[f64; 3], -) -> Array3 { - let (img_w, img_h, raw) = rgb_bytes(image); - let canvas_pixels = canvas_h * canvas_w; - - // Precompute fused scale/bias: (pixel/255 - mean) / std - let scale: [f32; 3] = std::array::from_fn(|c| 1.0 / (255.0 * std[c] as f32)); - let bias: [f32; 3] = std::array::from_fn(|c| -(mean[c] as f32) / (std[c] as f32)); - // Normalized value of black (0): 0 * scale + bias = bias - let pad_val = bias; - - let mut data = vec![0.0f32; 3 * canvas_pixels]; - let (r_plane, rest) = data.split_at_mut(canvas_pixels); - let (g_plane, b_plane) = rest.split_at_mut(canvas_pixels); - - // Pre-fill with padding value - r_plane.fill(pad_val[0]); - g_plane.fill(pad_val[1]); - b_plane.fill(pad_val[2]); - - // Overwrite image region row-by-row - let rw = img_w.min(canvas_w); - let rh = img_h.min(canvas_h); - for y in 0..rh { - let src_row = &raw[y * img_w * 3..y * img_w * 3 + rw * 3]; - let dst_offset = y * canvas_w; - let rd = &mut r_plane[dst_offset..dst_offset + rw]; - let gd = &mut g_plane[dst_offset..dst_offset + rw]; - let bd = &mut b_plane[dst_offset..dst_offset + rw]; - - for x in 0..rw { - let s = x * 3; - rd[x] = src_row[s] as f32 * scale[0] + bias[0]; - gd[x] = src_row[s + 1] as f32 * scale[1] + bias[1]; - bd[x] = src_row[s + 2] as f32 * scale[2] + bias[2]; - } - } - - #[expect( - clippy::expect_used, - reason = "data has exactly 3*canvas_h*canvas_w elements by construction" - )] - Array3::from_shape_vec((3, canvas_h, canvas_w), data) - .expect("shape matches pre-allocated buffer") -} - /// Calculate mean color of an image as RGB. pub fn calculate_mean_color(image: &DynamicImage) -> Rgb { let rgb = image.to_rgb8(); diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index 70a8fa998e..f4344aa334 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -407,18 +407,12 @@ async fn process_multimodal_parts( )); } - let image_sizes: Vec<(u32, u32)> = images - .iter() - .map(|f| (f.image.width(), f.image.height())) - .collect(); debug!( image_count = images.len(), - ?image_sizes, "Fetched images for multimodal processing" ); // Step 2: Resolve model spec and preprocess images - let t_config = std::time::Instant::now(); let model_config = components .get_or_load_config(model_id, tokenizer_source) .await?; @@ -440,23 +434,18 @@ async fn process_multimodal_parts( .image_processor_registry .find(model_id, model_type) .ok_or_else(|| anyhow::anyhow!("No image processor found for model: {model_id}"))?; - let config_us = t_config.elapsed().as_micros(); // ImagePreProcessor::preprocess takes &[DynamicImage]; images are behind Arc. // Clone cost is negligible vs. the preprocessing work itself. - let t_preprocess = std::time::Instant::now(); let dynamic_images: Vec = images.iter().map(|f| f.image.clone()).collect(); let preprocessed: PreprocessedImages = image_processor .preprocess(&dynamic_images, &model_config.preprocessor_config) .map_err(|e| anyhow::anyhow!("Image preprocessing failed: {e}"))?; - let preprocess_us = t_preprocess.elapsed().as_micros(); debug!( num_images = preprocessed.num_img_tokens.len(), total_tokens = preprocessed.num_img_tokens.iter().sum::(), - config_us, - preprocess_us, "Image preprocessing complete" ); From dc5b579b3280f2e20a2624ee66a3d711ffa1e5ed Mon Sep 17 00:00:00 2001 From: Chang Su Date: Wed, 1 Apr 2026 15:37:31 +0000 Subject: [PATCH 5/5] fix: lazily evaluate image_sizes in debug log Signed-off-by: Chang Su --- model_gateway/src/routers/grpc/multimodal.rs | 1 + 1 file changed, 1 insertion(+) diff --git a/model_gateway/src/routers/grpc/multimodal.rs b/model_gateway/src/routers/grpc/multimodal.rs index f4344aa334..a68ec14580 100644 --- a/model_gateway/src/routers/grpc/multimodal.rs +++ b/model_gateway/src/routers/grpc/multimodal.rs @@ -409,6 +409,7 @@ async fn process_multimodal_parts( debug!( image_count = images.len(), + image_sizes = ?images.iter().map(|f| (f.image.width(), f.image.height())).collect::>(), "Fetched images for multimodal processing" );