Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -123,6 +123,16 @@ async def test_render_to_generate_roundtrip(client, test_image):
assert "image" in features["kwargs_data"]
assert len(features["kwargs_data"]["image"]) > 0

assert "image_grid_thw" in features
assert "image" in features["image_grid_thw"]
image_grids = features["image_grid_thw"]["image"]
assert isinstance(image_grids, list)
assert len(image_grids) == len(features["kwargs_data"]["image"])
for grid in image_grids:
assert isinstance(grid, list)
assert len(grid) == 3
assert all(isinstance(v, int) and v > 0 for v in grid)

# Build generate request from render output
generate_payload = render_data
generate_payload["sampling_params"] = {
Expand Down
9 changes: 9 additions & 0 deletions vllm/entrypoints/serve/disagg/protocol.py
Original file line number Diff line number Diff line change
Expand Up @@ -58,6 +58,15 @@ class MultiModalFeatures(BaseModel):
``None`` for metadata-only (cache-hit) responses.
"""

image_grid_thw: dict[str, list[list[int] | None]] | None = None

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

high

The use of the | operator for type unions (e.g., list[int] | None) requires Python 3.10+ or from __future__ import annotations. Since vLLM supports Python 3.9, this will cause a TypeError at runtime. Please add from __future__ import annotations at the top of this file to maintain compatibility.

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

false positive
vLLM requires Python 3.10+ (pyproject.toml: requires-python = ">=3.10,<3.15")

"""Per-modality grid dimensions for mRoPE position computation.

Each value is a list parallel to ``mm_hashes[modality]``. Each entry
is a ``[grid_t, grid_h, grid_w]`` list (or ``None`` for cache hits).
This allows the prefill worker to compute mRoPE positions without
deserializing the full ``kwargs_data`` blobs.
"""


class GenerateRequest(BaseModel):
request_id: str = Field(
Expand Down
20 changes: 20 additions & 0 deletions vllm/entrypoints/serve/disagg/serving.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
import msgspec
import numpy as np
import pybase64 as base64
import torch
from fastapi import Request

from vllm.engine.protocol import EngineClient
Expand Down Expand Up @@ -43,6 +44,8 @@
from vllm.logger import init_logger
from vllm.logprobs import Logprob
from vllm.multimodal.inputs import (
MultiModalBatchedField,
MultiModalFieldElem,
MultiModalKwargsItem,
MultiModalKwargsItems,
PlaceholderRange,
Expand Down Expand Up @@ -156,6 +159,23 @@ async def serve_tokens(
decode_mm_kwargs_item(item) if item is not None else None
for item in items
]
elif features.image_grid_thw is not None:
# Lightweight path: construct minimal items containing
# only grid metadata for mRoPE position computation.
for modality, grids in features.image_grid_thw.items():
items_list: list[MultiModalKwargsItem | None] = []

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

high

The type hint list[MultiModalKwargsItem | None] uses the | union operator, which is not supported in Python 3.9 without from __future__ import annotations. Please add the future import at the top of this file.

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

false positive
vLLM requires Python 3.10+ (pyproject.toml: requires-python = ">=3.10,<3.15")

thw_key = f"{modality}_grid_thw"
for grid in grids:
if grid is not None:
tensor = torch.tensor(grid, dtype=torch.int64)
elem = MultiModalFieldElem(
data=tensor,
field=MultiModalBatchedField(keep_on_cpu=True),
)
items_list.append(MultiModalKwargsItem({thw_key: elem}))
else:
items_list.append(None)
mm_kwargs[modality] = items_list
else:
for modality, hashes in features.mm_hashes.items():
mm_kwargs[modality] = [None] * len(hashes)
Expand Down
20 changes: 19 additions & 1 deletion vllm/entrypoints/serve/render/serving.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,10 +2,13 @@
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
from collections.abc import Sequence
from http import HTTPStatus
from typing import Any, cast
from typing import TYPE_CHECKING, Any, cast

from openai_harmony import Message as OpenAIMessage

if TYPE_CHECKING:
import torch

from vllm.config import ModelConfig
from vllm.entrypoints.chat_utils import (
ChatTemplateContentFormatOption,
Expand Down Expand Up @@ -380,18 +383,33 @@ def _extract_mm_features(

# Serialize tensor data per modality.
kwargs_data: dict[str, list[str | None]] | None = None
image_grid_thw: dict[str, list[list[int] | None]] | None = None
if raw_mm_kwargs := mm_engine_input.get("mm_kwargs"):
kwargs_data = {}
image_grid_thw = {}
for modality, items in raw_mm_kwargs.items():
kwargs_data[modality] = [
encode_mm_kwargs_item(item) if item is not None else None
for item in items
]
thw_key = f"{modality}_grid_thw"
grids: list[list[int] | None] = []
for item in items:
if item is not None and thw_key in item:
thw_tensor = cast("torch.Tensor", item[thw_key].data)
grids.append(thw_tensor.tolist())

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

high

For models like Qwen2-VL, image_grid_thw is typically a tensor of shape (1, 3). Calling .tolist() on it returns a nested list [[t, h, w]], which violates the list[int] type expected by the protocol and will cause shape mismatches during reconstruction on the worker. Flatten the tensor before conversion.

Suggested change
grids.append(thw_tensor.tolist())
grids.append(thw_tensor.view(-1).tolist())

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Per-item image_grid_thw.data is already 1-D shape (3,), not (1, 3). Evidence: MultiModalBatchedField.build_elems splits the batched tensor along dim 0 (so (N, 3) → N elems of shape (3,)), and Qwen2-VL itself unpacks per-item with t, h, w = mm_feature.data["image_grid_thw"].data.tolist() (qwen2_vl.py:1217). So tolist() already returns a flat [t, h, w], which is exactly what your test asserts (len(grid) == 3, all ints). No .view(-1) needed

else:
grids.append(None)
if any(g is not None for g in grids):
image_grid_thw[modality] = grids
if not image_grid_thw:
image_grid_thw = None

return MultiModalFeatures(
mm_hashes=mm_hashes,
mm_placeholders=mm_placeholders,
kwargs_data=kwargs_data,
image_grid_thw=image_grid_thw,
)

def _make_request_with_harmony(
Expand Down
Loading