Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
22 commits
Select commit Hold shift + click to select a range
171d3aa
feat(multimodal): add blake3 image hasher module
CatherineSue Feb 28, 2026
8022cce
feat(proto): expand vLLM MultimodalInputs with preprocessed data fields
CatherineSue Feb 28, 2026
2e66130
feat(grpc): add mm_hashes to MultimodalData and expand into_vllm_proto
CatherineSue Feb 28, 2026
67f8bfb
feat(grpc): switch vLLM to preprocessed multimodal path with hashing
CatherineSue Feb 28, 2026
8c23931
feat(multimodal): add blake3 hash to ImageFrame at decode time
CatherineSue Feb 28, 2026
ba808a0
refactor(grpc): replace ProcessedMessages.multimodal_images with mult…
CatherineSue Feb 28, 2026
37cec93
refactor(grpc): collapse multimodal pipeline to single-phase in prepa…
CatherineSue Feb 28, 2026
86a5b96
style: rustfmt
CatherineSue Feb 28, 2026
c52abf7
feat(grpc): add batched_keys to MultimodalInputs proto and Rust pipeline
CatherineSue Feb 28, 2026
5ee072c
feat(grpc): expand TRT-LLM proto with preprocessed multimodal fields
CatherineSue Feb 28, 2026
8df1931
Revert "feat(grpc): expand TRT-LLM proto with preprocessed multimodal…
CatherineSue Feb 28, 2026
18f38a8
feat(grpc): add field_layouts and flat_keys to multimodal pipeline
CatherineSue Feb 28, 2026
542f887
feat(multimodal): add Qwen3-VL model processor spec
CatherineSue Feb 28, 2026
e6b6558
fix(multimodal): use int64 for patches_per_image to avoid torch.uint3…
CatherineSue Feb 28, 2026
632288d
feat(multimodal): structured prompt tokens for Llama4 and pass Prepro…
CatherineSue Mar 1, 2026
2c3ee5e
style: rustfmt
CatherineSue Mar 1, 2026
62b4846
refactor(proto): remove unused image_data from vLLM MultimodalInputs
CatherineSue Mar 1, 2026
c6b75ff
fix(multimodal): use chunks_exact and strict length check in extract_…
CatherineSue Mar 1, 2026
3595db9
nit: comments cleanup
CatherineSue Mar 1, 2026
0728a34
fix(grpc): emit patch-only placeholder offsets for sglang multimodal
CatherineSue Mar 1, 2026
2713a29
fix(multimodal): align QwenVL placeholder_token_id with pad_token_id
CatherineSue Mar 1, 2026
02fd148
refactor(grpc): compute sglang patch offsets during token expansion
CatherineSue Mar 1, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
41 changes: 39 additions & 2 deletions grpc_client/proto/vllm_engine.proto
Original file line number Diff line number Diff line change
Expand Up @@ -92,10 +92,47 @@ message TokenizedInput {
repeated uint32 input_ids = 2; // Actual token IDs to process
}

// A typed tensor: raw little-endian bytes + shape + dtype.
message TensorData {
bytes data = 1; // Raw little-endian bytes (f32/i64/u32)
repeated uint32 shape = 2; // Dimension sizes
string dtype = 3; // "float32", "int64", "uint32"
}

message PlaceholderRange {
uint32 offset = 1;
uint32 length = 2;
}

// Multimodal inputs for vision/audio models
message MultimodalInputs {
// Raw image bytes (JPEG/PNG) — vLLM handles preprocessing internally
repeated bytes image_data = 1;
// Reserved: field 1 was image_data (raw bytes), removed in favor of
// preprocessed pixel_values.
reserved 1;

// Preprocessed pixel values tensor
TensorData pixel_values = 2;

// Model-specific tensors (image_grid_thw, aspect_ratios, etc.)
map<string, TensorData> model_specific_tensors = 3;

// Image token ID used for placeholder expansion
optional uint32 im_token_id = 4;

// Placeholder offsets: where each image's tokens are in input_ids
repeated PlaceholderRange mm_placeholders = 5;

// Per-image blake3 hex hashes for encoder output caching
repeated string mm_hashes = 6;

// Tensor keys whose first dimension is per-image (batched).
// Keys not listed here are shared across all images.
repeated string batched_keys = 7;

// Tensor keys that need flat slicing: maps tensor name → sizes tensor name.
// e.g. {"pixel_values": "patches_per_image"} means pixel_values should be
// split per-item using sizes from the patches_per_image tensor.
map<string, string> flat_keys = 8;
}

// =====================
Expand Down
9 changes: 3 additions & 6 deletions model_gateway/src/routers/grpc/mod.rs
Original file line number Diff line number Diff line change
@@ -1,8 +1,5 @@
//! gRPC router implementations

use std::sync::Arc;

use llm_multimodal::ImageFrame;
use openai_protocol::common::StringOrArray;

pub mod client; // Used by core/
Expand All @@ -24,8 +21,8 @@ pub use proto_wrapper::{MultimodalData, TensorBytes};
#[derive(Debug)]
pub struct ProcessedMessages {
pub text: String,
/// Raw fetched images (Phase 1). Backend-specific preprocessing
/// happens in Phase 2 at request building time.
pub multimodal_images: Option<Vec<Arc<ImageFrame>>>,
/// Preprocessed multimodal data (pixel values, placeholders, hashes).
/// Populated during preparation when multimodal content is detected.
pub multimodal_data: Option<MultimodalData>,
pub stop_sequences: Option<StringOrArray>,
}
Loading