From d7de81a77311f1c62bdd5017e7920830a0306868 Mon Sep 17 00:00:00 2001 From: Harry Mellor <19981378+hmellor@users.noreply.github.com> Date: Tue, 18 Mar 2025 11:18:27 +0100 Subject: [PATCH 01/10] Replace `decord` with `torchcodec` Signed-off-by: Harry Mellor <19981378+hmellor@users.noreply.github.com> --- docs/source/conf.py | 1 - requirements/rocm-test.txt | 4 ---- requirements/test.in | 2 +- requirements/test.txt | 5 ++--- setup.py | 1 - vllm/multimodal/video.py | 14 +++++--------- 6 files changed, 8 insertions(+), 19 deletions(-) diff --git a/docs/source/conf.py b/docs/source/conf.py index b72faef9af10..4106464448f4 100644 --- a/docs/source/conf.py +++ b/docs/source/conf.py @@ -219,7 +219,6 @@ def linkcode_resolve(domain, info): "soundfile", "gguf", "lark", - "decord", ] for mock_target in autodoc_mock_imports: diff --git a/requirements/rocm-test.txt b/requirements/rocm-test.txt index 52fbf787f1df..0c2f40108035 100644 --- a/requirements/rocm-test.txt +++ b/requirements/rocm-test.txt @@ -12,10 +12,6 @@ soundfile==0.13.1 soxr==0.5.0.post1 librosa==0.10.2.post1 -# entrypoints test -#vllm[video] # required by entrypoints/openai/test_video.py -decord==0.6.0 - # entrypoints test #sentence-transformers # required by entrypoints/openai/test_score.py sentence-transformers==3.4.1 diff --git a/requirements/test.in b/requirements/test.in index faa4564eaa39..9c6d5939f88d 100644 --- a/requirements/test.in +++ b/requirements/test.in @@ -9,7 +9,6 @@ pytest-shard # testing utils awscli backoff # required for phi4mm test -decord # required for video tests einops # required for MPT, qwen-vl and Mamba httpx librosa # required for audio tests @@ -24,6 +23,7 @@ jiwer # required for audio tests timm # required for internvl test torch==2.6.0 torchaudio==2.6.0 +torchcodec==0.2.1 torchvision==0.21.0 transformers_stream_generator # required for qwen-vl test matplotlib # required for qwen-vl test diff --git a/requirements/test.txt b/requirements/test.txt index c733364fd871..58548256411f 100644 --- a/requirements/test.txt +++ b/requirements/test.txt @@ -92,8 +92,6 @@ datasets==3.0.2 # lm-eval decorator==5.1.1 # via librosa -decord==0.6.0 - # via -r requirements/test.in dill==0.3.8 # via # datasets @@ -271,7 +269,6 @@ numpy==1.26.4 # contourpy # cupy-cuda12x # datasets - # decord # einx # encodec # evaluate @@ -615,6 +612,8 @@ torchaudio==2.6.0 # -r requirements/test.in # encodec # vocos +torchcodec==0.2.1 + # via -r requirements/test.in torchvision==0.21.0 # via # -r requirements/test.in diff --git a/setup.py b/setup.py index d412f34b3e3d..cfdf98d7db8e 100755 --- a/setup.py +++ b/setup.py @@ -690,7 +690,6 @@ def _read_requirements(filename: str) -> list[str]: "tensorizer": ["tensorizer>=2.9.0"], "runai": ["runai-model-streamer", "runai-model-streamer-s3", "boto3"], "audio": ["librosa", "soundfile"], # Required for audio processing - "video": ["decord"] # Required for video processing }, cmdclass=cmdclass, package_data=package_data, diff --git a/vllm/multimodal/video.py b/vllm/multimodal/video.py index 0b3d3f8c79d7..de78641f5156 100644 --- a/vllm/multimodal/video.py +++ b/vllm/multimodal/video.py @@ -9,11 +9,12 @@ import numpy as np import numpy.typing as npt from PIL import Image +from torchcodec.decoders import VideoDecoder from vllm.inputs.registry import InputContext from vllm.logger import init_logger from vllm.transformers_utils.processor import cached_get_video_processor -from vllm.utils import PlaceholderModule, is_list_of +from vllm.utils import is_list_of from .base import MediaIO, ModalityData from .image import ImageMediaIO, ImagePlugin @@ -22,11 +23,6 @@ if TYPE_CHECKING: from vllm.config import ModelConfig -try: - import decord -except ImportError: - decord = PlaceholderModule("decord") # type: ignore[assignment] - logger = init_logger(__name__) @@ -131,8 +127,8 @@ def __init__( self.num_frames = num_frames def load_bytes(self, data: bytes) -> npt.NDArray: - vr = decord.VideoReader(BytesIO(data), num_threads=1) - total_frame_num = len(vr) + decoder = VideoDecoder(BytesIO(data)) + total_frame_num = len(decoder) num_frames = self.num_frames if total_frame_num > num_frames: @@ -144,7 +140,7 @@ def load_bytes(self, data: bytes) -> npt.NDArray: else: frame_idx = list(range(0, total_frame_num)) - return vr.get_batch(frame_idx).asnumpy() + return decoder.get_frames_at(frame_idx).data def load_base64(self, media_type: str, data: str) -> npt.NDArray: if media_type.lower() == "video/jpeg": From 3e7e67d4d53c3b1e4a4179343c214fdbad3063df Mon Sep 17 00:00:00 2001 From: Harry Mellor <19981378+hmellor@users.noreply.github.com> Date: Tue, 18 Mar 2025 12:31:00 +0100 Subject: [PATCH 02/10] Fix error Signed-off-by: Harry Mellor <19981378+hmellor@users.noreply.github.com> --- vllm/multimodal/video.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/vllm/multimodal/video.py b/vllm/multimodal/video.py index de78641f5156..0770e9e3951c 100644 --- a/vllm/multimodal/video.py +++ b/vllm/multimodal/video.py @@ -2,7 +2,6 @@ import base64 from functools import partial -from io import BytesIO from pathlib import Path from typing import TYPE_CHECKING, Any, Optional @@ -127,7 +126,7 @@ def __init__( self.num_frames = num_frames def load_bytes(self, data: bytes) -> npt.NDArray: - decoder = VideoDecoder(BytesIO(data)) + decoder = VideoDecoder(data) total_frame_num = len(decoder) num_frames = self.num_frames From 6e6536f96c7ee4ba80a494d393fc46b63bc80c0b Mon Sep 17 00:00:00 2001 From: Harry Mellor <19981378+hmellor@users.noreply.github.com> Date: Tue, 18 Mar 2025 12:31:29 +0100 Subject: [PATCH 03/10] Add `torchcodec` as a requirement for cuda and cpu Signed-off-by: Harry Mellor <19981378+hmellor@users.noreply.github.com> --- requirements/cpu.txt | 3 +++ requirements/cuda.txt | 1 + 2 files changed, 4 insertions(+) diff --git a/requirements/cpu.txt b/requirements/cpu.txt index b4e6abb6e3d6..96179e2cf515 100644 --- a/requirements/cpu.txt +++ b/requirements/cpu.txt @@ -10,6 +10,9 @@ torch==2.7.0.dev20250304; platform_machine == "s390x" torchaudio; platform_machine != "ppc64le" and platform_machine != "s390x" torchaudio==2.5.1; platform_machine == "ppc64le" +# required for video decoding, this must be updated alongside torch +torchcodec==0.2.1 + # required for the image processor of phi3v, this must be updated alongside torch torchvision; platform_machine != "ppc64le" and platform_machine != "s390x" torchvision==0.20.1; platform_machine == "ppc64le" diff --git a/requirements/cuda.txt b/requirements/cuda.txt index 702d4b0bb320..0aefd02875a3 100644 --- a/requirements/cuda.txt +++ b/requirements/cuda.txt @@ -8,5 +8,6 @@ ray[cgraph]>=2.43.0 # Ray Compiled Graph, required for pipeline parallelism in V torch==2.6.0 torchaudio==2.6.0 # These must be updated alongside torch +torchcodec==0.2.1 # Required for video decoding torchvision==0.21.0 # Required for phi3v processor. See https://github.com/pytorch/vision?tab=readme-ov-file#installation for corresponding version xformers==0.0.29.post2; platform_system == 'Linux' and platform_machine == 'x86_64' # Requires PyTorch 2.6.0 From f3604945025c8bfca01503ebc72530f7b4039346 Mon Sep 17 00:00:00 2001 From: Harry Mellor <19981378+hmellor@users.noreply.github.com> Date: Tue, 18 Mar 2025 12:47:29 +0100 Subject: [PATCH 04/10] Fix docs build Signed-off-by: Harry Mellor <19981378+hmellor@users.noreply.github.com> --- docs/source/conf.py | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/source/conf.py b/docs/source/conf.py index 4106464448f4..a366aab7fe42 100644 --- a/docs/source/conf.py +++ b/docs/source/conf.py @@ -219,6 +219,7 @@ def linkcode_resolve(domain, info): "soundfile", "gguf", "lark", + "torchcodec", ] for mock_target in autodoc_mock_imports: From 24918900040d33f13cf0572d2dfaf5c4e5cbf73f Mon Sep 17 00:00:00 2001 From: Harry Mellor <19981378+hmellor@users.noreply.github.com> Date: Tue, 18 Mar 2025 12:48:07 +0100 Subject: [PATCH 05/10] Output video as np array Signed-off-by: Harry Mellor <19981378+hmellor@users.noreply.github.com> --- vllm/multimodal/video.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/vllm/multimodal/video.py b/vllm/multimodal/video.py index 0770e9e3951c..18fbee8c4010 100644 --- a/vllm/multimodal/video.py +++ b/vllm/multimodal/video.py @@ -139,7 +139,7 @@ def load_bytes(self, data: bytes) -> npt.NDArray: else: frame_idx = list(range(0, total_frame_num)) - return decoder.get_frames_at(frame_idx).data + return decoder.get_frames_at(frame_idx).data.numpy() def load_base64(self, media_type: str, data: str) -> npt.NDArray: if media_type.lower() == "video/jpeg": From e86b3c749a6c26987a0192225523ea88e3e6a012 Mon Sep 17 00:00:00 2001 From: Harry Mellor <19981378+hmellor@users.noreply.github.com> Date: Tue, 18 Mar 2025 14:27:47 +0100 Subject: [PATCH 06/10] Fix `VideoDecoder` instantiation Signed-off-by: Harry Mellor <19981378+hmellor@users.noreply.github.com> --- vllm/multimodal/video.py | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/vllm/multimodal/video.py b/vllm/multimodal/video.py index 18fbee8c4010..bd84d991153f 100644 --- a/vllm/multimodal/video.py +++ b/vllm/multimodal/video.py @@ -126,16 +126,17 @@ def __init__( self.num_frames = num_frames def load_bytes(self, data: bytes) -> npt.NDArray: - decoder = VideoDecoder(data) - total_frame_num = len(decoder) + decoder = VideoDecoder(data, + dimension_order="NHWC", + seek_mode="approximate") num_frames = self.num_frames + total_frame_num = len(decoder) if total_frame_num > num_frames: - uniform_sampled_frames = np.linspace(0, - total_frame_num - 1, - num_frames, - dtype=int) - frame_idx = uniform_sampled_frames.tolist() + frame_idx = np.linspace(0, + total_frame_num - 1, + num_frames, + dtype=int).tolist() else: frame_idx = list(range(0, total_frame_num)) From 38a429226f47053b91bec66c18cf9c18a7c0a326 Mon Sep 17 00:00:00 2001 From: Harry Mellor <19981378+hmellor@users.noreply.github.com> Date: Tue, 18 Mar 2025 14:32:37 +0100 Subject: [PATCH 07/10] Add `torchcodec` to rocm requirements Signed-off-by: Harry Mellor <19981378+hmellor@users.noreply.github.com> --- requirements/rocm-build.txt | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/requirements/rocm-build.txt b/requirements/rocm-build.txt index a0731c51d46b..77424265a171 100644 --- a/requirements/rocm-build.txt +++ b/requirements/rocm-build.txt @@ -3,8 +3,9 @@ --extra-index-url https://download.pytorch.org/whl/rocm6.2 torch==2.5.1 -torchvision==0.20.1 torchaudio==2.5.1 +torchcodec==0.1.1 +torchvision==0.20.1 cmake>=3.26 packaging From fa63f52745f3961934e6ecb26a6c4d2ae30c2ef1 Mon Sep 17 00:00:00 2001 From: Harry Mellor <19981378+hmellor@users.noreply.github.com> Date: Wed, 19 Mar 2025 14:36:35 +0100 Subject: [PATCH 08/10] Respond to comment Signed-off-by: Harry Mellor <19981378+hmellor@users.noreply.github.com> --- setup.py | 1 + 1 file changed, 1 insertion(+) diff --git a/setup.py b/setup.py index cfdf98d7db8e..19a42299d64f 100755 --- a/setup.py +++ b/setup.py @@ -690,6 +690,7 @@ def _read_requirements(filename: str) -> list[str]: "tensorizer": ["tensorizer>=2.9.0"], "runai": ["runai-model-streamer", "runai-model-streamer-s3", "boto3"], "audio": ["librosa", "soundfile"], # Required for audio processing + "video": [], # Does nothing, kept for compatibility }, cmdclass=cmdclass, package_data=package_data, From 75312925963e6523b33fa475decff38fa9c9391e Mon Sep 17 00:00:00 2001 From: Harry Mellor <19981378+hmellor@users.noreply.github.com> Date: Wed, 19 Mar 2025 13:45:23 +0000 Subject: [PATCH 09/10] Apply suggestions from code review Signed-off-by: Harry Mellor <19981378+hmellor@users.noreply.github.com> Co-authored-by: Isotr0py <2037008807@qq.com> --- requirements/cpu.txt | 2 +- requirements/cuda.txt | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/requirements/cpu.txt b/requirements/cpu.txt index 96179e2cf515..746c3f6c254e 100644 --- a/requirements/cpu.txt +++ b/requirements/cpu.txt @@ -11,7 +11,7 @@ torchaudio; platform_machine != "ppc64le" and platform_machine != "s390x" torchaudio==2.5.1; platform_machine == "ppc64le" # required for video decoding, this must be updated alongside torch -torchcodec==0.2.1 +torchcodec==0.2.1; platform_machine == "x86_64" or platform_system == "Darwin" # required for the image processor of phi3v, this must be updated alongside torch torchvision; platform_machine != "ppc64le" and platform_machine != "s390x" diff --git a/requirements/cuda.txt b/requirements/cuda.txt index 0aefd02875a3..aa0830c7a3de 100644 --- a/requirements/cuda.txt +++ b/requirements/cuda.txt @@ -8,6 +8,6 @@ ray[cgraph]>=2.43.0 # Ray Compiled Graph, required for pipeline parallelism in V torch==2.6.0 torchaudio==2.6.0 # These must be updated alongside torch -torchcodec==0.2.1 # Required for video decoding +torchcodec==0.2.1; platform_system == 'Linux' and platform_machine == 'x86_64' # Required for video decoding torchvision==0.21.0 # Required for phi3v processor. See https://github.com/pytorch/vision?tab=readme-ov-file#installation for corresponding version xformers==0.0.29.post2; platform_system == 'Linux' and platform_machine == 'x86_64' # Requires PyTorch 2.6.0 From 1e35f4422e19f344fb337d3ec8ecc627f9975dde Mon Sep 17 00:00:00 2001 From: Harry Mellor <19981378+hmellor@users.noreply.github.com> Date: Thu, 20 Mar 2025 17:10:03 +0100 Subject: [PATCH 10/10] Respond to comment Signed-off-by: Harry Mellor <19981378+hmellor@users.noreply.github.com> --- requirements/rocm-build.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/requirements/rocm-build.txt b/requirements/rocm-build.txt index 77424265a171..5b5266dabd4c 100644 --- a/requirements/rocm-build.txt +++ b/requirements/rocm-build.txt @@ -4,7 +4,7 @@ --extra-index-url https://download.pytorch.org/whl/rocm6.2 torch==2.5.1 torchaudio==2.5.1 -torchcodec==0.1.1 +torchcodec==0.1.1; platform_machine == "x86_64" torchvision==0.20.1 cmake>=3.26