-
Notifications
You must be signed in to change notification settings - Fork 178
feat(gateway): add multimodal support to Messages API gRPC pipeline #776
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -1,23 +1,29 @@ | ||
| //! Multimodal processing integration for gRPC chat pipeline. | ||
| //! Multimodal processing integration for gRPC pipeline (chat + messages). | ||
| //! | ||
| //! This module bridges the `llm-multimodal` crate with the gRPC router pipeline, | ||
| //! handling the full processing chain: extract content parts → fetch images → | ||
| //! preprocess pixels → expand placeholder tokens → build proto MultimodalInputs. | ||
| //! | ||
| //! Both the chat completion pipeline and the Messages API pipeline share the same | ||
| //! processing core (`process_multimodal_parts`). Only the detection and extraction | ||
| //! functions differ because they work with different input types (`ChatMessage` vs | ||
| //! `InputMessage`). | ||
|
|
||
| use std::{collections::HashMap, path::Path, sync::Arc}; | ||
|
|
||
| use anyhow::{Context, Result}; | ||
| use dashmap::DashMap; | ||
| use llm_multimodal::{ | ||
| AsyncMultiModalTracker, ChatContentPart, FieldLayout, ImageDetail, ImageFrame, | ||
| ImageProcessorRegistry, MediaConnector, MediaConnectorConfig, Modality, ModelMetadata, | ||
| ModelRegistry, ModelSpecificValue, PlaceholderRange, PreProcessorConfig, PreprocessedImages, | ||
| AsyncMultiModalTracker, FieldLayout, ImageDetail, ImageFrame, ImageProcessorRegistry, | ||
| MediaConnector, MediaConnectorConfig, MediaContentPart, Modality, ModelMetadata, ModelRegistry, | ||
| ModelSpecificValue, PlaceholderRange, PreProcessorConfig, PreprocessedImages, | ||
| PromptReplacement, TrackedMedia, TrackerOutput, | ||
| }; | ||
| use llm_tokenizer::TokenizerTrait; | ||
| use openai_protocol::{ | ||
| chat::{ChatMessage, MessageContent}, | ||
| common::ContentPart, | ||
| messages::{ImageSource, InputContent, InputContentBlock, InputMessage, Role}, | ||
| }; | ||
| use tracing::{debug, warn}; | ||
|
|
||
|
|
@@ -165,8 +171,8 @@ pub(crate) fn has_multimodal_content(messages: &[ChatMessage]) -> bool { | |
| } | ||
|
|
||
| /// Extract multimodal content parts from OpenAI chat messages, | ||
| /// converting protocol `ContentPart` to multimodal crate `ChatContentPart`. | ||
| fn extract_content_parts(messages: &[ChatMessage]) -> Vec<ChatContentPart> { | ||
| /// converting protocol `ContentPart` to multimodal crate `MediaContentPart`. | ||
| fn extract_content_parts(messages: &[ChatMessage]) -> Vec<MediaContentPart> { | ||
| let mut parts = Vec::new(); | ||
|
|
||
| for msg in messages { | ||
|
|
@@ -182,14 +188,14 @@ fn extract_content_parts(messages: &[ChatMessage]) -> Vec<ChatContentPart> { | |
| match part { | ||
| ContentPart::ImageUrl { image_url } => { | ||
| let detail = image_url.detail.as_deref().and_then(parse_detail); | ||
| parts.push(ChatContentPart::ImageUrl { | ||
| parts.push(MediaContentPart::ImageUrl { | ||
| url: image_url.url.clone(), | ||
| detail, | ||
| uuid: None, | ||
| }); | ||
| } | ||
| ContentPart::Text { text } => { | ||
| parts.push(ChatContentPart::Text { text: text.clone() }); | ||
| parts.push(MediaContentPart::Text { text: text.clone() }); | ||
| } | ||
| ContentPart::VideoUrl { .. } => {} // Skip VideoUrl for now | ||
| } | ||
|
|
@@ -210,6 +216,96 @@ fn parse_detail(detail: &str) -> Option<ImageDetail> { | |
| } | ||
| } | ||
|
|
||
| // --------------------------------------------------------------------------- | ||
| // Messages API multimodal detection and extraction | ||
| // --------------------------------------------------------------------------- | ||
|
|
||
| /// Check if any messages in a Messages API request contain multimodal content. | ||
| pub(crate) fn has_multimodal_content_messages(messages: &[InputMessage]) -> bool { | ||
| messages.iter().any(|msg| { | ||
| if msg.role != Role::User { | ||
| return false; | ||
| } | ||
| match &msg.content { | ||
| InputContent::Blocks(blocks) => blocks | ||
| .iter() | ||
| .any(|block| matches!(block, InputContentBlock::Image(_))), | ||
| InputContent::String(_) => false, | ||
| } | ||
| }) | ||
|
Comment on lines
+225
to
+235
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. The closure passed to messages.iter().any(|msg| {
msg.role == Role::User
&& match &msg.content {
InputContent::Blocks(blocks) => blocks
.iter()
.any(|block| matches!(block, InputContentBlock::Image(_))),
InputContent::String(_) => false,
}
})References
Member
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. This comment is valid |
||
| } | ||
|
coderabbitai[bot] marked this conversation as resolved.
|
||
|
|
||
| /// Extract multimodal content parts from Messages API input messages, | ||
| /// converting `InputContentBlock::Image` to multimodal crate `MediaContentPart`. | ||
| fn extract_content_parts_messages(messages: &[InputMessage]) -> Vec<MediaContentPart> { | ||
| let mut parts = Vec::new(); | ||
|
|
||
| for msg in messages { | ||
| if msg.role != Role::User { | ||
| continue; | ||
| } | ||
| let blocks = match &msg.content { | ||
| InputContent::Blocks(blocks) => blocks, | ||
| InputContent::String(_) => continue, | ||
| }; | ||
|
|
||
| for block in blocks { | ||
| match block { | ||
| InputContentBlock::Image(image_block) => match &image_block.source { | ||
| ImageSource::Base64 { media_type, data } => { | ||
| // Convert base64 to data URL for the media connector | ||
| let data_url = format!("data:{media_type};base64,{data}"); | ||
| parts.push(MediaContentPart::ImageUrl { | ||
| url: data_url, | ||
| detail: None, | ||
| uuid: None, | ||
| }); | ||
| } | ||
| ImageSource::Url { url } => { | ||
| parts.push(MediaContentPart::ImageUrl { | ||
| url: url.clone(), | ||
| detail: None, | ||
| uuid: None, | ||
| }); | ||
| } | ||
| }, | ||
| InputContentBlock::Text(text_block) => { | ||
| parts.push(MediaContentPart::Text { | ||
| text: text_block.text.clone(), | ||
| }); | ||
| } | ||
| _ => {} | ||
| } | ||
| } | ||
| } | ||
|
|
||
| parts | ||
| } | ||
|
Comment on lines
+238
to
+283
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. 🧹 Nitpick | 🔵 Trivial Add a brief comment explaining which block types are intentionally skipped. The 📝 Suggested comment InputContentBlock::Text(text_block) => {
parts.push(MediaContentPart::Text {
text: text_block.text.clone(),
});
}
- _ => {}
+ // Skip Document, ToolUse, ToolResult, etc. — only images and text are
+ // processed for multimodal; other block types pass through unchanged.
+ _ => {}
}🤖 Prompt for AI Agents |
||
|
|
||
| /// Process multimodal content from Messages API input messages. | ||
| /// | ||
| /// Entry point for the messages preparation stage. Extracts image content parts | ||
| /// from `InputMessage`, then delegates to the shared processing core. | ||
| pub(crate) async fn process_multimodal_messages( | ||
| messages: &[InputMessage], | ||
| model_id: &str, | ||
| tokenizer: &dyn TokenizerTrait, | ||
| token_ids: Vec<u32>, | ||
| components: &MultimodalComponents, | ||
| tokenizer_source: &str, | ||
| ) -> Result<MultimodalOutput> { | ||
| let content_parts = extract_content_parts_messages(messages); | ||
| process_multimodal_parts( | ||
| content_parts, | ||
| model_id, | ||
| tokenizer, | ||
| token_ids, | ||
| components, | ||
| tokenizer_source, | ||
| ) | ||
| .await | ||
| } | ||
|
|
||
| /// Process multimodal content: fetch images, preprocess pixels, expand tokens, collect hashes. | ||
| /// | ||
| /// Single entry point called from preparation.rs. Handles the full pipeline: | ||
|
|
@@ -222,8 +318,30 @@ pub(crate) async fn process_multimodal( | |
| components: &MultimodalComponents, | ||
| tokenizer_source: &str, | ||
| ) -> Result<MultimodalOutput> { | ||
| // Step 1: Fetch images | ||
| let content_parts = extract_content_parts(messages); | ||
| process_multimodal_parts( | ||
| content_parts, | ||
| model_id, | ||
| tokenizer, | ||
| token_ids, | ||
| components, | ||
| tokenizer_source, | ||
| ) | ||
| .await | ||
| } | ||
|
|
||
| /// Shared multimodal processing core. | ||
| /// | ||
| /// Takes pre-extracted `MediaContentPart`s (from either chat or messages pipeline) | ||
| /// and runs the full processing chain: fetch → preprocess → expand → build intermediate. | ||
| async fn process_multimodal_parts( | ||
| content_parts: Vec<MediaContentPart>, | ||
| model_id: &str, | ||
| tokenizer: &dyn TokenizerTrait, | ||
| token_ids: Vec<u32>, | ||
| components: &MultimodalComponents, | ||
| tokenizer_source: &str, | ||
| ) -> Result<MultimodalOutput> { | ||
| let mut tracker = AsyncMultiModalTracker::new(components.media_connector.clone()); | ||
|
Member
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Let's add comment originally at https://github.com/lightseekorg/smg/pull/776/changes#diff-a50caf12eac40e563302f834a904db895c131b4dbbade4eeb5dc61c9be6db9dbL225 here? |
||
|
|
||
| for part in content_parts { | ||
|
|
@@ -674,12 +792,12 @@ mod tests { | |
| assert_eq!(parts.len(), 2); | ||
|
|
||
| match &parts[0] { | ||
| ChatContentPart::Text { text } => assert_eq!(text, "Describe this:"), | ||
| MediaContentPart::Text { text } => assert_eq!(text, "Describe this:"), | ||
| _ => panic!("Expected Text part"), | ||
| } | ||
|
|
||
| match &parts[1] { | ||
| ChatContentPart::ImageUrl { url, detail, .. } => { | ||
| MediaContentPart::ImageUrl { url, detail, .. } => { | ||
| assert_eq!(url, "https://example.com/image.jpg"); | ||
| assert_eq!(*detail, Some(ImageDetail::High)); | ||
| } | ||
|
|
||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
🧹 Nitpick | 🔵 Trivial
Keep a compatibility export for the renamed public type.
Dropping
ChatContentPartfrom the crate root will break downstream imports immediately. Ifllm-multimodalis consumed outside this workspace, please either keep a deprecated alias for one release or ship this behind an explicit breaking-version bump.Possible compatibility shim
pub use types::{ MediaContentPart, FieldLayout, ImageDetail, ImageFrame, ImageSize, ImageSource, Modality, MultiModalData, MultiModalUUIDs, PlaceholderRange, PromptReplacement, TokenId, TrackedMedia, }; +#[deprecated(note = "renamed to MediaContentPart")] +pub use types::MediaContentPart as ChatContentPart;📝 Committable suggestion
🤖 Prompt for AI Agents