Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion docs/source/conf.py
Original file line number Diff line number Diff line change
Expand Up @@ -219,7 +219,7 @@ def linkcode_resolve(domain, info):
"soundfile",
"gguf",
"lark",
"decord",
"torchcodec",
]

for mock_target in autodoc_mock_imports:
Expand Down
3 changes: 3 additions & 0 deletions requirements/cpu.txt
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,9 @@ torch==2.7.0.dev20250304; platform_machine == "s390x"
torchaudio; platform_machine != "ppc64le" and platform_machine != "s390x"
torchaudio==2.5.1; platform_machine == "ppc64le"

# required for video decoding, this must be updated alongside torch
torchcodec==0.2.1; platform_machine == "x86_64" or platform_system == "Darwin"

# required for the image processor of phi3v, this must be updated alongside torch
torchvision; platform_machine != "ppc64le" and platform_machine != "s390x"
torchvision==0.20.1; platform_machine == "ppc64le"
Expand Down
1 change: 1 addition & 0 deletions requirements/cuda.txt
Original file line number Diff line number Diff line change
Expand Up @@ -8,5 +8,6 @@ ray[cgraph]>=2.43.0 # Ray Compiled Graph, required for pipeline parallelism in V
torch==2.6.0
torchaudio==2.6.0
# These must be updated alongside torch
torchcodec==0.2.1; platform_system == 'Linux' and platform_machine == 'x86_64' # Required for video decoding
torchvision==0.21.0 # Required for phi3v processor. See https://github.com/pytorch/vision?tab=readme-ov-file#installation for corresponding version
xformers==0.0.29.post2; platform_system == 'Linux' and platform_machine == 'x86_64' # Requires PyTorch 2.6.0
3 changes: 2 additions & 1 deletion requirements/rocm-build.txt
Original file line number Diff line number Diff line change
Expand Up @@ -3,8 +3,9 @@

--extra-index-url https://download.pytorch.org/whl/rocm6.2
torch==2.5.1
torchvision==0.20.1
torchaudio==2.5.1
torchcodec==0.1.1; platform_machine == "x86_64"
torchvision==0.20.1

cmake>=3.26
packaging
Expand Down
4 changes: 0 additions & 4 deletions requirements/rocm-test.txt
Original file line number Diff line number Diff line change
Expand Up @@ -12,10 +12,6 @@ soundfile==0.13.1
soxr==0.5.0.post1
librosa==0.10.2.post1

# entrypoints test
#vllm[video] # required by entrypoints/openai/test_video.py
decord==0.6.0

# entrypoints test
#sentence-transformers # required by entrypoints/openai/test_score.py
sentence-transformers==3.4.1
Expand Down
2 changes: 1 addition & 1 deletion requirements/test.in
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,6 @@ pytest-shard
# testing utils
awscli
backoff # required for phi4mm test
decord # required for video tests
einops # required for MPT, qwen-vl and Mamba
httpx
librosa # required for audio tests
Expand All @@ -24,6 +23,7 @@ jiwer # required for audio tests
timm # required for internvl test
torch==2.6.0
torchaudio==2.6.0
torchcodec==0.2.1
torchvision==0.21.0
transformers_stream_generator # required for qwen-vl test
matplotlib # required for qwen-vl test
Expand Down
5 changes: 2 additions & 3 deletions requirements/test.txt
Original file line number Diff line number Diff line change
Expand Up @@ -92,8 +92,6 @@ datasets==3.0.2
# lm-eval
decorator==5.1.1
# via librosa
decord==0.6.0
# via -r requirements/test.in
dill==0.3.8
# via
# datasets
Expand Down Expand Up @@ -271,7 +269,6 @@ numpy==1.26.4
# contourpy
# cupy-cuda12x
# datasets
# decord
# einx
# encodec
# evaluate
Expand Down Expand Up @@ -615,6 +612,8 @@ torchaudio==2.6.0
# -r requirements/test.in
# encodec
# vocos
torchcodec==0.2.1
# via -r requirements/test.in
torchvision==0.21.0
# via
# -r requirements/test.in
Expand Down
2 changes: 1 addition & 1 deletion setup.py
Original file line number Diff line number Diff line change
Expand Up @@ -690,7 +690,7 @@ def _read_requirements(filename: str) -> list[str]:
"tensorizer": ["tensorizer>=2.9.0"],
"runai": ["runai-model-streamer", "runai-model-streamer-s3", "boto3"],
"audio": ["librosa", "soundfile"], # Required for audio processing
"video": ["decord"] # Required for video processing
"video": [], # Does nothing, kept for compatibility
},
Comment thread
hmellor marked this conversation as resolved.
cmdclass=cmdclass,
package_data=package_data,
Expand Down
26 changes: 11 additions & 15 deletions vllm/multimodal/video.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,18 +2,18 @@

import base64
from functools import partial
from io import BytesIO
from pathlib import Path
from typing import TYPE_CHECKING, Any, Optional

import numpy as np
import numpy.typing as npt
from PIL import Image
from torchcodec.decoders import VideoDecoder

from vllm.inputs.registry import InputContext
from vllm.logger import init_logger
from vllm.transformers_utils.processor import cached_get_video_processor
from vllm.utils import PlaceholderModule, is_list_of
from vllm.utils import is_list_of

from .base import MediaIO, ModalityData
from .image import ImageMediaIO, ImagePlugin
Expand All @@ -22,11 +22,6 @@
if TYPE_CHECKING:
from vllm.config import ModelConfig

try:
import decord
except ImportError:
decord = PlaceholderModule("decord") # type: ignore[assignment]

logger = init_logger(__name__)


Expand Down Expand Up @@ -131,20 +126,21 @@ def __init__(
self.num_frames = num_frames

def load_bytes(self, data: bytes) -> npt.NDArray:
vr = decord.VideoReader(BytesIO(data), num_threads=1)
total_frame_num = len(vr)
decoder = VideoDecoder(data,
dimension_order="NHWC",
seek_mode="approximate")

num_frames = self.num_frames
total_frame_num = len(decoder)
if total_frame_num > num_frames:
uniform_sampled_frames = np.linspace(0,
total_frame_num - 1,
num_frames,
dtype=int)
frame_idx = uniform_sampled_frames.tolist()
frame_idx = np.linspace(0,
total_frame_num - 1,
num_frames,
dtype=int).tolist()
else:
frame_idx = list(range(0, total_frame_num))

return vr.get_batch(frame_idx).asnumpy()
return decoder.get_frames_at(frame_idx).data.numpy()

def load_base64(self, media_type: str, data: str) -> npt.NDArray:
if media_type.lower() == "video/jpeg":
Expand Down