Skip to content
Merged
Changes from 1 commit
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 6 additions & 2 deletions vllm/model_executor/models/qwen3_omni_moe_thinker.py
Original file line number Diff line number Diff line change
Expand Up @@ -711,11 +711,12 @@ def pad_to_hop_length(x: np.ndarray, hop_length: int) -> np.ndarray:
return x

# NOTE: WhisperFeatureExtractor cannot handle empty list of audios
feature_extractor = self.info.get_feature_extractor()

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Can you add a comment so we can revert once it's fixed on transformers side?

hop_length = feature_extractor.hop_length
if audios:
# NOTE: Qwen3-Omni processor accept "audio"
# To make sure the cache works with padding=True, we pre-padded
# the audio to multiple of hop_length.
hop_length = self.info.get_feature_extractor().hop_length
mm_data["audio"] = [
pad_to_hop_length(audio, hop_length)
if isinstance(audio, np.ndarray)
Expand All @@ -738,7 +739,6 @@ def pad_to_hop_length(x: np.ndarray, hop_length: int) -> np.ndarray:
and "feature_attention_mask" in hf_inputs
and (audios := mm_data.get("audio", []))
):
hop_length = self.info.get_feature_extractor().hop_length
audio_num_frames = []
for _, audio in enumerate(audios):
audio_length = len(audio[0]) if isinstance(audio, tuple) else len(audio)
Expand All @@ -747,6 +747,10 @@ def pad_to_hop_length(x: np.ndarray, hop_length: int) -> np.ndarray:
if audio_length % hop_length == 0
else (audio_length // hop_length - 1)
)
if mm_kwargs.get("truncation", True):
num_frame = min(
num_frame, feature_extractor.n_samples // hop_length
)
audio_num_frames.append(num_frame)
hf_inputs["feature_attention_mask"] = [
torch.ones(num_frame) for num_frame in audio_num_frames
Expand Down
Loading