Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions crates/multimodal/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,7 @@ name = "llm_multimodal"
base64 = "0.22"
hf-hub = "0.5.0"
bytes = { version = "1.8.0", features = ["serde"] }
fast_image_resize = { version = "6.0.0", features = ["image"] }
image = { version = "0.25.4", default-features = false, features = ["png", "jpeg", "gif", "bmp", "ico", "tiff", "webp"] }
ndarray = "0.17"
once_cell = "1.21.3"
Expand All @@ -36,9 +37,14 @@ anyhow.workspace = true
blake3.workspace = true

[dev-dependencies]
criterion = { version = "0.5", features = ["html_reports"] }
npyz = { version = "0.8", features = ["npz"] }
tempfile = "3.8"
tokio = { workspace = true, features = ["rt-multi-thread", "macros"] }

[[bench]]
name = "image_preprocess"
harness = false

[lints]
workspace = true
368 changes: 368 additions & 0 deletions crates/multimodal/benches/image_preprocess.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,368 @@
//! Benchmark: SMG image preprocessing vs HF processor baseline.
//!
//! Measures the time for model-specific image preprocessing (resize, normalize,
//! patchify) at various image sizes. Compare results with the companion Python
//! script `scripts/bench_image_preprocess.py` which benchmarks HF transformers.
//!
//! Run: cargo bench -p llm-multimodal --bench image_preprocess

#![allow(clippy::unwrap_used, clippy::expect_used)]

use criterion::{criterion_group, criterion_main, BenchmarkId, Criterion};
use image::{imageops::FilterType, DynamicImage, RgbImage};
use llm_multimodal::vision::{
image_processor::ImagePreProcessor,
preprocessor_config::PreProcessorConfig,
processors::{Llama4VisionProcessor, Qwen2VLProcessor, Qwen3VLProcessor},
transforms,
};

/// Create a synthetic RGB image with some variation (not all zeros).
fn make_test_image(width: u32, height: u32) -> DynamicImage {
let img = RgbImage::from_fn(width, height, |x, y| {
image::Rgb([(x % 256) as u8, (y % 256) as u8, ((x + y) % 256) as u8])
});
DynamicImage::ImageRgb8(img)
}

fn load_preprocessor_config(model_path: &str) -> Option<PreProcessorConfig> {
let config_path = format!("{model_path}/preprocessor_config.json");
let json = std::fs::read_to_string(&config_path).ok()?;
PreProcessorConfig::from_json(&json).ok()
}

// ── Full pipeline benchmarks ─────────────────────────────────────

fn bench_qwen3_vl(c: &mut Criterion) {
let processor = Qwen3VLProcessor::new();
let config =
load_preprocessor_config("/raid/models/Qwen/Qwen3-VL-8B-Instruct").unwrap_or_else(|| {
PreProcessorConfig::from_json(
r#"{"do_resize": true, "size": {"shortest_edge": 3136, "longest_edge": 12845056}}"#,
)
.unwrap()
});

let sizes: &[(u32, u32)] = &[
(224, 224),
(640, 480),
(1024, 768),
(1920, 1080),
(3840, 2160),
];

let mut group = c.benchmark_group("qwen3_vl_preprocess");
for &(w, h) in sizes {
let image = make_test_image(w, h);
let images = [image];
group.bench_with_input(
BenchmarkId::new("single", format!("{w}x{h}")),
&images,
|b, imgs| {
b.iter(|| processor.preprocess(imgs, &config).unwrap());
},
);
}
group.finish();

// Batch benchmarks
let mut group = c.benchmark_group("qwen3_vl_batch");
for batch_size in [3, 5, 10] {
let images: Vec<DynamicImage> = (0..batch_size)
.map(|i| make_test_image(640 + i * 10, 480 + i * 10))
.collect();
group.bench_with_input(
BenchmarkId::new("640x480", format!("batch{batch_size}")),
&images,
|b, imgs| {
b.iter(|| processor.preprocess(imgs, &config).unwrap());
},
);
}
group.finish();

// Extreme: very small and very large
let mut group = c.benchmark_group("qwen3_vl_extreme");
let extremes: &[(u32, u32, &str)] = &[
(32, 32, "tiny_32x32"),
(50, 50, "small_50x50"),
(100, 2000, "tall_100x2000"),
(2000, 100, "wide_2000x100"),
(4096, 4096, "huge_4096x4096"),
];
for &(w, h, label) in extremes {
let image = make_test_image(w, h);
let images = [image];
group.bench_with_input(BenchmarkId::new("single", label), &images, |b, imgs| {
b.iter(|| processor.preprocess(imgs, &config).unwrap());
});
}
group.finish();
}

fn bench_qwen2_vl(c: &mut Criterion) {
let processor = Qwen2VLProcessor::new();
let config =
load_preprocessor_config("/raid/models/Qwen/Qwen2-VL-2B-Instruct").unwrap_or_else(|| {
PreProcessorConfig::from_json(
r#"{"do_resize": true, "size": {"shortest_edge": 3136, "longest_edge": 12845056}}"#,
)
.unwrap()
});

let sizes: &[(u32, u32)] = &[(224, 224), (640, 480), (1024, 768), (1920, 1080)];

let mut group = c.benchmark_group("qwen2_vl_preprocess");
for &(w, h) in sizes {
let image = make_test_image(w, h);
let images = [image];
group.bench_with_input(
BenchmarkId::new("single", format!("{w}x{h}")),
&images,
|b, imgs| {
b.iter(|| processor.preprocess(imgs, &config).unwrap());
},
);
}
group.finish();
}

fn bench_llama4(c: &mut Criterion) {
let processor = Llama4VisionProcessor::new();
let config =
load_preprocessor_config("/raid/models/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8")
.unwrap_or_else(|| {
PreProcessorConfig::from_json(
r#"{"do_resize": true, "size": {"height": 336, "width": 336}}"#,
)
.unwrap()
});

let sizes: &[(u32, u32)] = &[
(224, 224),
(336, 336),
(640, 480),
(1024, 768),
(1920, 1080),
];

let mut group = c.benchmark_group("llama4_preprocess");
for &(w, h) in sizes {
let image = make_test_image(w, h);
let images = [image];
group.bench_with_input(
BenchmarkId::new("single", format!("{w}x{h}")),
&images,
|b, imgs| {
b.iter(|| processor.preprocess(imgs, &config).unwrap());
},
);
}
group.finish();
}

// ── Per-step profiling benchmarks ────────────────────────────────

fn bench_individual_steps(c: &mut Criterion) {
let sizes: &[(u32, u32)] = &[(640, 480), (1024, 768), (1920, 1080)];

// Step 1: Resize only
let mut group = c.benchmark_group("step_resize");
for &(w, h) in sizes {
let image = make_test_image(w, h);
// Qwen3-VL target: smart_resize result
let processor = Qwen3VLProcessor::new();
let (th, tw) = processor.smart_resize(h as usize, w as usize).unwrap();
group.bench_with_input(
BenchmarkId::new("fir_bilinear", format!("{w}x{h}")),
&image,
|b, img| {
b.iter(|| transforms::resize(img, tw as u32, th as u32, FilterType::Triangle));
},
);
}
group.finish();

// Step 2: to_tensor only
let mut group = c.benchmark_group("step_to_tensor");
for &(w, h) in sizes {
// Use a pre-resized image to isolate to_tensor cost
let image = make_test_image(w, h);
group.bench_with_input(
BenchmarkId::new("rgb8", format!("{w}x{h}")),
&image,
|b, img| {
b.iter(|| transforms::to_tensor(img));
},
);
}
group.finish();

// Step 3: normalize only
let mut group = c.benchmark_group("step_normalize");
let mean = [0.5, 0.5, 0.5];
let std = [0.5, 0.5, 0.5];
for &(w, h) in sizes {
let image = make_test_image(w, h);
let tensor = transforms::to_tensor(&image);
group.bench_with_input(
BenchmarkId::new("f32", format!("{w}x{h}")),
&tensor,
|b, t| {
b.iter_batched(
|| t.clone(),
|mut fresh| transforms::normalize(&mut fresh, &mean, &std),
criterion::BatchSize::SmallInput,
);
},
Comment thread
coderabbitai[bot] marked this conversation as resolved.
);
}
group.finish();
}

fn bench_llama4_steps(c: &mut Criterion) {
let processor = Llama4VisionProcessor::new();
let config =
load_preprocessor_config("/raid/models/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8")
.unwrap_or_else(|| {
PreProcessorConfig::from_json(
r#"{"do_resize": true, "size": {"height": 336, "width": 336}}"#,
)
.unwrap()
});

// 1024x768 is the worst case (1.8x slower than HF)
let sizes: &[(u32, u32)] = &[(640, 480), (1024, 768), (1920, 1080)];

let mut group = c.benchmark_group("llama4_steps");
for &(w, h) in sizes {
let image = make_test_image(w, h);

// Full preprocess
group.bench_with_input(
BenchmarkId::new("full_preprocess", format!("{w}x{h}")),
&image,
|b, img| {
let imgs = [img.clone()];
b.iter(|| processor.preprocess(&imgs, &config).unwrap());
},
);

// Just resize
group.bench_with_input(
BenchmarkId::new("resize_only", format!("{w}x{h}")),
&image,
|b, img| {
b.iter(|| transforms::resize(img, 336, 336, FilterType::Triangle));
},
);

// to_tensor_and_normalize on tile-sized image
let tile_img = make_test_image(336, 336);
group.bench_with_input(
BenchmarkId::new("tensor_normalize_336", format!("{w}x{h}")),
&tile_img,
|b, img| {
b.iter(|| {
transforms::to_tensor_and_normalize(img, &[0.5, 0.5, 0.5], &[0.5, 0.5, 0.5])
});
},
);
Comment thread
CatherineSue marked this conversation as resolved.
}
group.finish();
}

criterion_group!(
benches,
bench_qwen3_vl,
bench_qwen2_vl,
bench_llama4,
bench_llama4_steps,
bench_individual_steps,
bench_fused_to_tensor_normalize,
bench_to_rgb8,
bench_resize_detailed,
Comment thread
CatherineSue marked this conversation as resolved.
);
criterion_main!(benches);

fn bench_fused_to_tensor_normalize(c: &mut Criterion) {
let sizes: &[(u32, u32)] = &[(640, 480), (1024, 768), (1920, 1080)];
let mean = [0.5, 0.5, 0.5];
let std = [0.5, 0.5, 0.5];

let mut group = c.benchmark_group("step_to_tensor_normalize_fused");
for &(w, h) in sizes {
let image = make_test_image(w, h);
group.bench_with_input(
BenchmarkId::new("fused", format!("{w}x{h}")),
&image,
|b, img| {
b.iter(|| transforms::to_tensor_and_normalize(img, &mean, &std));
},
);
}
group.finish();
}

fn bench_to_rgb8(c: &mut Criterion) {
let sizes: &[(u32, u32)] = &[(640, 480), (1024, 768), (1920, 1080)];

let mut group = c.benchmark_group("step_to_rgb8");
for &(w, h) in sizes {
let image = make_test_image(w, h);
group.bench_with_input(
BenchmarkId::new("rgb8", format!("{w}x{h}")),
&image,
|b, img| {
b.iter(|| img.to_rgb8());
},
);
}
group.finish();

// Also test when image is already RGB8 (should be free)
let mut group = c.benchmark_group("step_to_rgb8_noop");
for &(w, h) in sizes {
let image = make_test_image(w, h);
let rgb = DynamicImage::ImageRgb8(image.to_rgb8());
group.bench_with_input(
BenchmarkId::new("already_rgb8", format!("{w}x{h}")),
&rgb,
|b, img| {
b.iter(|| img.to_rgb8());
},
Comment on lines +307 to +333

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue | 🟠 Major

🧩 Analysis chain

🏁 Script executed:

# First, let's look at the full file to understand context
cd crates/multimodal && wc -l benches/image_preprocess.rs

Repository: lightseekorg/smg

Length of output: 91


🏁 Script executed:

# Look at the bench_to_rgb8 function and surrounding code
cd crates/multimodal && sed -n '280,350p' benches/image_preprocess.rs

Repository: lightseekorg/smg

Length of output: 2328


🏁 Script executed:

# Search for make_test_image function definition
cd crates/multimodal && grep -n "fn make_test_image" benches/image_preprocess.rs

Repository: lightseekorg/smg

Length of output: 124


🏁 Script executed:

# Read the make_test_image function implementation
cd crates/multimodal && sed -n '21,35p' benches/image_preprocess.rs

Repository: lightseekorg/smg

Length of output: 628


🏁 Script executed:

# Look for imports and Cargo.toml to understand what image crate is used
cd crates/multimodal && head -30 benches/image_preprocess.rs

Repository: lightseekorg/smg

Length of output: 1358


🌐 Web query:

image crate DynamicImage to_rgb8() implementation behavior

💡 Result:

The DynamicImage::to_rgb8(&self) -> RgbImage method returns a copy of the image converted to an 8-bit-per-channel RGB format (ImageBuffer<Rgb, Vec>). It works by pattern-matching on the DynamicImage enum variants and calling convert on each underlying ImageBuffer using the ConvertBuffer trait. This performs pixel-by-pixel color space conversion: - From RGB8: identity copy (no change). - From RGBA8: discards alpha channel, copies RGB channels directly. - From grayscale (Luma8): expands to RGB using luminance-to-RGB conversion (typically sRGB coefficients: R=G=B = 0.299R + 0.587G + 0.114B, but since grayscale, replicates the luma value). - From grayscale-alpha (LumaA8): similar to Luma8, ignoring alpha. - From higher bit depths (16-bit, float): downsamples to u8 and converts color space. - WARNING from docs: RGB↔Luma conversions always use fixed sRGB coefficients, ignoring image color space. The method always allocates a new RgbImage buffer. It operates in encoded (non-linear) color space without internal linearization. Example usage: use image::{DynamicImage, open}; let img: DynamicImage = open("input.png")?; let rgb = img.to_rgb8;

Citations:


🏁 Script executed:

# Also check if there's any documentation or comments in the file about to_rgb8
cd crates/multimodal && grep -n "to_rgb8" benches/image_preprocess.rs -B2 -A2

Repository: lightseekorg/smg

Length of output: 1116


Both benchmark groups measure the same RGB8 allocation/copy path, not a fast-path optimization.

make_test_image() returns DynamicImage::ImageRgb8, so step_to_rgb8 benchmarks to_rgb8() on an already-RGB8 image. The step_to_rgb8_noop group is also not a noop: DynamicImage::to_rgb8() always allocates a new buffer, even when converting from RGB8 to RGB8 (it performs an "identity copy"). The comment "should be free" is incorrect—calling to_rgb8() on an ImageRgb8-wrapped image still incurs allocation overhead. Both benchmarks measure the same operation, making the results misleading for claims about fast-path performance.

🤖 Prompt for AI Agents
Verify each finding against the current code and only fix it if needed.

In `@crates/multimodal/benches/image_preprocess.rs` around lines 304 - 330, The
two benchmark groups are both calling DynamicImage::to_rgb8(), which always
allocates a new buffer, so the "noop" group is not measuring a fast-path; update
the "step_to_rgb8_noop" group to benchmark the true no-copy fast-path by using
DynamicImage::as_rgb8() (or directly using the inner ImageBuffer created by
make_test_image) instead of to_rgb8(), e.g., create rgb =
DynamicImage::ImageRgb8(image.to_rgb8()) and in the bench closure call
img.as_rgb8() (or operate on the ImageBuffer directly) and update the comment to
no longer claim "should be free" but to state that as_rgb8() is the
non-allocating path; alternatively, add a separate benchmark that converts other
DynamicImage variants (e.g., ImageRgba8 or ImageLuma8) with to_rgb8() to measure
real conversion cost.

);
}
group.finish();
}

fn bench_resize_detailed(c: &mut Criterion) {
let processor = Qwen3VLProcessor::new();

// Benchmark: make_test_image + resize + convert back to DynamicImage
let mut group = c.benchmark_group("step_resize_full_pipeline");
let sizes: &[(u32, u32)] = &[(640, 480), (1024, 768), (1920, 1080)];
for &(w, h) in sizes {
let image = make_test_image(w, h);
let (th, tw) = processor.smart_resize(h as usize, w as usize).unwrap();

// fir resize (our path)
group.bench_with_input(
BenchmarkId::new("fir", format!("{w}x{h}->{tw}x{th}")),
&image,
|b, img| {
b.iter(|| transforms::resize(img, tw as u32, th as u32, FilterType::Triangle));
},
);

// image crate resize (old path, for comparison)
group.bench_with_input(
BenchmarkId::new("image_crate", format!("{w}x{h}->{tw}x{th}")),
&image,
|b, img| {
b.iter(|| img.resize_exact(tw as u32, th as u32, FilterType::Triangle));
},
);
}
group.finish();
}
Loading