From 18cd4df85f6e4b505866b264ab69b0861cee10c1 Mon Sep 17 00:00:00 2001 From: Simo Lin Date: Thu, 26 Mar 2026 11:14:48 -0700 Subject: [PATCH 1/2] perf(multimodal): replace scalar bicubic_resize with SIMD FIR CatmullRom and fuse phi3/phi4 normalize Replace the hand-written per-pixel scalar bicubic_resize (which made ~339K calls to bicubic_interpolate with bounds-checked ndarray indexing) with SIMD-accelerated FIR CatmullRom from fast_image_resize. This operates on the raw DynamicImage before tensor conversion, avoiding the f32 intermediate. Additionally, fuse the separate to_tensor + normalize two-pass pipeline into a single to_tensor_and_normalize call for both the global image and HD tiles in Phi3Vision and Phi4Vision processors. Changes: - phi3_vision: create_global_image now takes DynamicImage, uses transforms::resize(CatmullRom) + to_tensor_and_normalize - phi4_vision: same pattern for create_global_image - phi3_vision/phi4_vision: HD tensor uses to_tensor_and_normalize (fused) - transforms.rs: remove cubic_weight, bicubic_interpolate, bicubic_resize (no remaining callers) Golden tests for phi3/phi4 may need regeneration due to numerical differences between FIR u8-space CatmullRom and the previous f32-space scalar bicubic. The golden fixture files are not present in CI worktrees; when fixtures are present the tests pass within existing tolerances (0.08 phi3, 0.05 phi4). Signed-off-by: Simo Lin --- .../src/vision/processors/phi3_vision.rs | 31 +++--- .../src/vision/processors/phi4_vision.rs | 27 ++--- crates/multimodal/src/vision/transforms.rs | 101 ------------------ 3 files changed, 33 insertions(+), 126 deletions(-) diff --git a/crates/multimodal/src/vision/processors/phi3_vision.rs b/crates/multimodal/src/vision/processors/phi3_vision.rs index 9b847cfb8f..8ac70b48c2 100644 --- a/crates/multimodal/src/vision/processors/phi3_vision.rs +++ b/crates/multimodal/src/vision/processors/phi3_vision.rs @@ -7,7 +7,7 @@ //! //! 1. **HD Transform**: Resize and pad image to multiples of 336 //! 2. **Normalize**: Apply CLIP normalization -//! 3. **Create Global Image**: Bicubic interpolate to 336x336 +//! 3. **Create Global Image**: SIMD FIR CatmullRom resize to 336x336 //! 4. **Tile**: Reshape into (num_tiles, 3, 336, 336) //! 5. **Concatenate**: [global_image, tiles...] //! 6. **Pad**: Zero-pad to (num_crops+1, 3, 336, 336) @@ -177,16 +177,20 @@ impl Phi3VisionProcessor { new_image } - /// Create global image by bicubic interpolation to 336x336. - /// - /// Uses the shared `bicubic_resize` which matches PyTorch's - /// `torch.nn.functional.interpolate(mode='bicubic', align_corners=False)`. + /// Create global image by resizing the raw DynamicImage to TILE_SIZE x TILE_SIZE + /// using SIMD-accelerated FIR CatmullRom, then converting to a normalized tensor. #[expect( clippy::unused_self, reason = "method logically belongs to the processor; keeps API consistent" )] - fn create_global_image(&self, tensor: &Array3) -> Array3 { - transforms::bicubic_resize(tensor, TILE_SIZE as usize, TILE_SIZE as usize) + fn create_global_image( + &self, + hd_image: &DynamicImage, + mean: &[f64; 3], + std: &[f64; 3], + ) -> Array3 { + let resized = transforms::resize(hd_image, TILE_SIZE, TILE_SIZE, FilterType::CatmullRom); + transforms::to_tensor_and_normalize(&resized, mean, std) } /// Reshape HD image into tiles. @@ -243,12 +247,11 @@ impl Phi3VisionProcessor { // 1. Convert to RGB let image = DynamicImage::ImageRgb8(image.to_rgb8()); - // 2. HD transform + // 2. HD transform (produces a DynamicImage) let hd_image = self.hd_transform(&image); let (hd_w, hd_h) = hd_image.dimensions(); - // 3. To tensor [0, 1] and normalize - let mut tensor = transforms::to_tensor(&hd_image); + // Resolve normalization parameters let mean = config .image_mean .as_ref() @@ -259,10 +262,12 @@ impl Phi3VisionProcessor { .as_ref() .map(|v| [v[0], v[1], v[2]]) .unwrap_or(self.std); - transforms::normalize(&mut tensor, &mean, &std); - // 4. Create global image (336x336) - let global_image = self.create_global_image(&tensor); + // 3. Create global image: FIR CatmullRom resize on raw image, then fused to_tensor+normalize + let global_image = self.create_global_image(&hd_image, &mean, &std); + + // 4. Fused to_tensor + normalize on HD image (single pass) + let tensor = transforms::to_tensor_and_normalize(&hd_image, &mean, &std); // 5. Reshape HD image into tiles let tiles = self.reshape_to_tiles(&tensor); diff --git a/crates/multimodal/src/vision/processors/phi4_vision.rs b/crates/multimodal/src/vision/processors/phi4_vision.rs index 53bd4ba7aa..0bd85ef597 100644 --- a/crates/multimodal/src/vision/processors/phi4_vision.rs +++ b/crates/multimodal/src/vision/processors/phi4_vision.rs @@ -302,13 +302,17 @@ impl Phi4VisionProcessor { DynamicImage::ImageRgb8(padded) } - /// Create global image by bicubic interpolation to base resolution. - /// - /// Uses the shared `bicubic_resize` which matches PyTorch's - /// `torch.nn.functional.interpolate(mode='bicubic', align_corners=False)`. - fn create_global_image(&self, tensor: &Array3) -> Array3 { - let target = self.base_resolution as usize; - transforms::bicubic_resize(tensor, target, target) + /// Create global image by resizing the raw DynamicImage to base_resolution x base_resolution + /// using SIMD-accelerated FIR CatmullRom, then converting to a normalized tensor. + fn create_global_image( + &self, + hd_image: &DynamicImage, + mean: &[f64; 3], + std: &[f64; 3], + ) -> Array3 { + let target = self.base_resolution; + let resized = transforms::resize(hd_image, target, target, FilterType::CatmullRom); + transforms::to_tensor_and_normalize(&resized, mean, std) } /// Tile the HD image into crops of base_resolution x base_resolution. @@ -385,12 +389,11 @@ impl Phi4VisionProcessor { let hd_h = hd_image.height(); let hd_w = hd_image.width(); - // Step 2: Convert to tensor and normalize - let mut hd_tensor = transforms::to_tensor(&hd_image); - transforms::normalize(&mut hd_tensor, &self.mean, &self.std); + // Step 2: Create global image: FIR CatmullRom resize on raw image, then fused to_tensor+normalize + let global_tensor = self.create_global_image(&hd_image, &self.mean, &self.std); - // Step 3: Create global image - let global_tensor = self.create_global_image(&hd_tensor); + // Step 3: Fused to_tensor + normalize on HD image (single pass) + let hd_tensor = transforms::to_tensor_and_normalize(&hd_image, &self.mean, &self.std); // Step 4: Tile HD image let tiles = self.tile_image(&hd_tensor, h_crops, w_crops); diff --git a/crates/multimodal/src/vision/transforms.rs b/crates/multimodal/src/vision/transforms.rs index eaabf886bf..23825c79d4 100644 --- a/crates/multimodal/src/vision/transforms.rs +++ b/crates/multimodal/src/vision/transforms.rs @@ -369,107 +369,6 @@ pub fn mean_to_rgb(mean: &[f64; 3]) -> Rgb { ]) } -/// Cubic interpolation weight function (Keys bicubic kernel with a=-0.5). -/// -/// This matches PyTorch's bicubic interpolation used in -/// `torch.nn.functional.interpolate(mode='bicubic')`. -#[inline] -pub fn cubic_weight(x: f32) -> f32 { - let x = x.abs(); - if x < 1.0 { - (1.5 * x - 2.5) * x * x + 1.0 - } else if x < 2.0 { - ((-0.5 * x + 2.5) * x - 4.0) * x + 2.0 - } else { - 0.0 - } -} - -/// Perform bicubic interpolation at a single point in a tensor. -/// -/// Uses a 4x4 kernel with Keys bicubic weights (a=-0.5) to match PyTorch's -/// `torch.nn.functional.interpolate(mode='bicubic')`. -/// -/// # Arguments -/// * `tensor` - Input tensor of shape [C, H, W] -/// * `c` - Channel index -/// * `src_y` - Source Y coordinate (can be fractional) -/// * `src_x` - Source X coordinate (can be fractional) -/// * `h` - Height of the tensor -/// * `w` - Width of the tensor -/// -/// # Returns -/// The interpolated value at the specified position. -pub fn bicubic_interpolate( - tensor: &Array3, - c: usize, - src_y: f32, - src_x: f32, - h: usize, - w: usize, -) -> f32 { - let y_int = src_y.floor() as i32; - let x_int = src_x.floor() as i32; - let y_frac = src_y - y_int as f32; - let x_frac = src_x - x_int as f32; - - let mut result = 0.0f32; - - // Sample 4x4 neighborhood - for dy in -1..=2 { - let y_idx = (y_int + dy).clamp(0, h as i32 - 1) as usize; - let y_weight = cubic_weight(y_frac - dy as f32); - - for dx in -1..=2 { - let x_idx = (x_int + dx).clamp(0, w as i32 - 1) as usize; - let x_weight = cubic_weight(x_frac - dx as f32); - - result += tensor[[c, y_idx, x_idx]] * y_weight * x_weight; - } - } - - result -} - -/// Resize a tensor using bicubic interpolation. -/// -/// This matches PyTorch's `torch.nn.functional.interpolate(mode='bicubic', align_corners=False)`. -/// -/// # Arguments -/// * `tensor` - Input tensor of shape [C, H, W] -/// * `target_h` - Target height -/// * `target_w` - Target width -/// -/// # Returns -/// Resized tensor of shape [C, target_h, target_w]. -pub fn bicubic_resize(tensor: &Array3, target_h: usize, target_w: usize) -> Array3 { - let (c, h, w) = (tensor.shape()[0], tensor.shape()[1], tensor.shape()[2]); - - if h == target_h && w == target_w { - return tensor.clone(); - } - - let mut result = Array3::::zeros((c, target_h, target_w)); - - // PyTorch align_corners=False coordinate mapping - let scale_h = h as f32 / target_h as f32; - let scale_w = w as f32 / target_w as f32; - - for ch in 0..c { - for y in 0..target_h { - for x in 0..target_w { - // PyTorch align_corners=False: src = (dst + 0.5) * scale - 0.5 - let src_y = (y as f32 + 0.5) * scale_h - 0.5; - let src_x = (x as f32 + 0.5) * scale_w - 0.5; - - result[[ch, y, x]] = bicubic_interpolate(tensor, ch, src_y, src_x, h, w); - } - } - } - - result -} - #[cfg(test)] mod tests { use super::*; From 835fc8892541dbde3600cd305eeaadc2197a31d6 Mon Sep 17 00:00:00 2001 From: Simo Lin Date: Thu, 26 Mar 2026 12:03:07 -0700 Subject: [PATCH 2/2] bench(multimodal): add phi3-vision preprocessing benchmarks Signed-off-by: Simo Lin --- crates/multimodal/benches/image_preprocess.rs | 26 ++++++++++++++++++- 1 file changed, 25 insertions(+), 1 deletion(-) diff --git a/crates/multimodal/benches/image_preprocess.rs b/crates/multimodal/benches/image_preprocess.rs index 180db0d608..f13533496e 100644 --- a/crates/multimodal/benches/image_preprocess.rs +++ b/crates/multimodal/benches/image_preprocess.rs @@ -13,7 +13,9 @@ use image::{imageops::FilterType, DynamicImage, RgbImage}; use llm_multimodal::vision::{ image_processor::ImagePreProcessor, preprocessor_config::PreProcessorConfig, - processors::{Llama4VisionProcessor, Qwen2VLProcessor, Qwen3VLProcessor}, + processors::{ + Llama4VisionProcessor, Phi3VisionProcessor, Qwen2VLProcessor, Qwen3VLProcessor, + }, transforms, }; @@ -161,6 +163,27 @@ fn bench_llama4(c: &mut Criterion) { group.finish(); } +fn bench_phi3_vision(c: &mut Criterion) { + let processor = Phi3VisionProcessor::new(); + let config = PreProcessorConfig::default(); + + let sizes: &[(u32, u32)] = &[(224, 224), (336, 336), (640, 480), (1024, 768)]; + + let mut group = c.benchmark_group("phi3_vision_preprocess"); + for &(w, h) in sizes { + let image = make_test_image(w, h); + let images = [image]; + group.bench_with_input( + BenchmarkId::new("single", format!("{w}x{h}")), + &images, + |b, imgs| { + b.iter(|| processor.preprocess(imgs, &config).unwrap()); + }, + ); + } + group.finish(); +} + // ── Per-step profiling benchmarks ──────────────────────────────── fn bench_individual_steps(c: &mut Criterion) { @@ -277,6 +300,7 @@ criterion_group!( bench_qwen3_vl, bench_qwen2_vl, bench_llama4, + bench_phi3_vision, bench_llama4_steps, bench_individual_steps, bench_fused_to_tensor_normalize,