diff --git a/apps/ComfyUI-vLLM-Omni/README.md b/apps/ComfyUI-vLLM-Omni/README.md index c26b2a3de50..1d34faa54c5 100644 --- a/apps/ComfyUI-vLLM-Omni/README.md +++ b/apps/ComfyUI-vLLM-Omni/README.md @@ -271,10 +271,10 @@ The [WF-05 template](example_workflows/vLLM-Omni%20MiniMax-H3%20Latent%20Mask%20 Connect a **Latent Mask Editing** node to **Generate Video → latent_edit** to edit a source clip instead of generating from scratch. It uploads the source media and serializes the video/audio noise masks the MiniMax H3 API accepts: - `source_video` / `source_audio` — the media to edit. -- `video_mask` — a ComfyUI mask image; `0` preserves a region, `1` regenerates it, fractional values blend. A 2D mask `[H, W]` is applied to every frame; a 3D mask `[T, H, W]` is treated as a temporal mask (one slice per frame) for continuation or extension. It is resized to the video latent grid. +- `video_mask` — a ComfyUI mask image; `0` preserves a region, `1` regenerates it, fractional values blend. A 2D mask `[H, W]` is applied to every frame; a 3D mask `[T, H, W]` is treated as a temporal mask (one slice per frame) for continuation or extension. The client area-downsamples it by 16 to keep the upload small, and the server resizes it to the video latent grid. - `audio_mask` — a scalar in `[0, 1]`; `0` keeps the source audio, `1` regenerates it, fractional values blend. -A non-trivial mask requires its matching source, and a source without a mask is rejected. This node only forwards inputs to the server; the served model must declare latent-mask editing support (MiniMax-H3). +The node requires at least one mask; the server then enforces the cross-field rules (a source requires a matching mask, and a non-trivial mask requires its source). This node only forwards inputs to the server; the served model must declare latent-mask editing support (MiniMax-H3). ### TTS (e.g., Qwen TTS series) diff --git a/apps/ComfyUI-vLLM-Omni/__init__.py b/apps/ComfyUI-vLLM-Omni/__init__.py index 999f85ab26b..a59a5f0e4b6 100644 --- a/apps/ComfyUI-vLLM-Omni/__init__.py +++ b/apps/ComfyUI-vLLM-Omni/__init__.py @@ -36,7 +36,6 @@ # A dictionary that contains all nodes you want to export with their names NODE_CLASS_MAPPINGS = { - "VLLMOmniMiniMaxH3TemporalMask": VLLMOmniMiniMaxH3TemporalMask, # === Generation === "VLLMOmniGenerateImage": VLLMOmniGenerateImage, "VLLMOmniGenerateVideo": VLLMOmniGenerateVideo, @@ -46,6 +45,7 @@ "VLLMOmniVoiceClone": VLLMOmniVoiceClone, "VLLMOmniVideoReferences": VLLMOmniVideoReferences, "VLLMOmniLatentMaskEditing": VLLMOmniLatentMaskEditing, + "VLLMOmniMiniMaxH3TemporalMask": VLLMOmniMiniMaxH3TemporalMask, # === Params === "VLLMOmniARSampling": VLLMOmniARSampling, "VLLMOmniDiffusionSampling": VLLMOmniDiffusionSampling, @@ -59,7 +59,6 @@ # A dictionary that contains the friendly/humanly readable titles for the nodes NODE_DISPLAY_NAME_MAPPINGS = { - "VLLMOmniMiniMaxH3TemporalMask": "MiniMax-H3 Temporal Mask", # === Generation === "VLLMOmniGenerateImage": "Generate Image", "VLLMOmniGenerateVideo": "Generate Video", @@ -69,6 +68,7 @@ "VLLMOmniVoiceClone": "TTS Voice Cloning", "VLLMOmniVideoReferences": "Video References", "VLLMOmniLatentMaskEditing": "Latent Mask Editing", + "VLLMOmniMiniMaxH3TemporalMask": "MiniMax-H3 Temporal Mask", # === Params === "VLLMOmniARSampling": "AR Sampling Params", "VLLMOmniDiffusionSampling": "Diffusion Sampling Params", diff --git a/apps/ComfyUI-vLLM-Omni/comfyui_vllm_omni/nodes.py b/apps/ComfyUI-vLLM-Omni/comfyui_vllm_omni/nodes.py index 5bddb934508..96066fdc4b7 100644 --- a/apps/ComfyUI-vLLM-Omni/comfyui_vllm_omni/nodes.py +++ b/apps/ComfyUI-vLLM-Omni/comfyui_vllm_omni/nodes.py @@ -8,7 +8,7 @@ from comfy_api.input import AudioInput, VideoInput from .utils.api_client import VLLMOmniClient -from .utils.latent_mask import _align_frame_count, _video_latent_t +from .utils.latent_mask import _align_frame_count from .utils.logger import get_logger from .utils.models import lookup_model_spec from .utils.types import ( @@ -1133,13 +1133,13 @@ def build(self, images, source_fps, duration, mode, preserve_fraction=0.5): else: raise ValueError(f"Unknown temporal mask mode: {mode}") available = min(frames, math.floor(boundary * 24 + 1e-8)) + # Snap the preserved prefix to whole VAE clips (5 + 17n frames) so every + # latent it maps to is fully preserved. prefix = 0 if available < 5 else 5 + 17 * ((available - 5) // 17) - preserved = _video_latent_t(prefix) if prefix else 0 - total = _video_latent_t(frames) - mask = torch.ones(total, 1, 1) - mask[:preserved] = 0 + # One slice per output frame; the server pools frames to the latent grid. + mask = torch.ones(frames, 1, 1) + mask[:prefix] = 0 indices = torch.arange(frames, device=images.device) source_indices = (indices * (source_fps / 24)).floor().long().clamp(max=images.shape[0] - 1) preview_images = images.index_select(0, source_indices) - preview_mask = mask.index_select(0, torch.arange(frames) * total // frames) - return mask, 24.0, preview_images, preview_mask + return mask, 24.0, preview_images, mask diff --git a/apps/ComfyUI-vLLM-Omni/comfyui_vllm_omni/utils/api_client.py b/apps/ComfyUI-vLLM-Omni/comfyui_vllm_omni/utils/api_client.py index 69e1af89948..fff87859c59 100644 --- a/apps/ComfyUI-vLLM-Omni/comfyui_vllm_omni/utils/api_client.py +++ b/apps/ComfyUI-vLLM-Omni/comfyui_vllm_omni/utils/api_client.py @@ -30,7 +30,7 @@ video_to_base64, video_to_bytes, ) -from .latent_mask import scalar_mask_to_json, video_mask_to_grid_json +from .latent_mask import scalar_mask_to_json, video_mask_to_json from .logger import get_logger, pretty_printer from .models import lookup_model_spec from .types import ( @@ -388,6 +388,15 @@ async def generate_video( content_type="image/png", ) + if len(keyframe_images) == 2: + for image_filename, image in keyframe_images: + form.add_field( + "input_references", + image_tensor_to_png_bytes(image, image_filename), + filename=image_filename, + content_type="image/png", + ) + # === latent-mask editing (MiniMax H3) === if latent_edit is not None: source_video = latent_edit.get("source_video") @@ -398,21 +407,16 @@ async def generate_video( if video_mask is None and audio_mask is None: raise ValueError("Latent-mask editing requires at least one mask.") - video_mask_trivial = video_mask is None or bool((video_mask == 1.0).all().item()) - audio_mask_trivial = audio_mask is None or audio_mask == 1.0 - if not video_mask_trivial and source_video is None: - raise ValueError("A non-trivial video mask requires a source video.") - if not audio_mask_trivial and source_audio is None and source_video is None: - raise ValueError("A non-trivial audio mask requires a source audio or a source video with audio.") - if source_video is not None: form.add_field( "source_video", - video_to_bytes(source_video, "source.mp4"), + video_to_bytes(source_video), filename="source.mp4", content_type="video/mp4", ) if source_audio is not None: + # The filename extension selects the audio codec (mp3), so it is + # meaningful even though aiohttp uses the explicit filename below. form.add_field( "source_audio", audio_to_bytes(source_audio, "source_audio.mp3"), @@ -420,7 +424,7 @@ async def generate_video( content_type="audio/mpeg", ) if video_mask is not None: - mask_json = video_mask_to_grid_json(video_mask, width=width, height=height, num_frames=num_frames) + mask_json = video_mask_to_json(video_mask) form.add_field( "video_noise_mask", mask_json.encode("utf-8"), @@ -435,15 +439,6 @@ async def generate_video( content_type="application/json", ) - if len(keyframe_images) == 2: - for image_filename, image in keyframe_images: - form.add_field( - "input_references", - image_tensor_to_png_bytes(image, image_filename), - filename=image_filename, - content_type="image/png", - ) - # === model specific params. Either use a specialized builder, or add flattened fields as-is === if model_params is not None: model_params = dict(model_params) diff --git a/apps/ComfyUI-vLLM-Omni/comfyui_vllm_omni/utils/latent_mask.py b/apps/ComfyUI-vLLM-Omni/comfyui_vllm_omni/utils/latent_mask.py index 532cf818406..07820fd0115 100644 --- a/apps/ComfyUI-vLLM-Omni/comfyui_vllm_omni/utils/latent_mask.py +++ b/apps/ComfyUI-vLLM-Omni/comfyui_vllm_omni/utils/latent_mask.py @@ -1,11 +1,25 @@ # SPDX-License-Identifier: Apache-2.0 # SPDX-FileCopyrightText: Copyright contributors to the vLLM-Omni project +"""Client-side serialization of MiniMax-H3 latent-edit masks. + +A video mask is sent in frame space — a 2D spatial ``[H, W]`` (applied to every +frame) or a 3D ``[T, H, W]`` (one slice per frame) — and the server resolves it +to the latent grid. An audio mask is a scalar applied to all time steps. +""" + import json +import math import torch import torch.nn.functional as F +# The server rejects mask JSON above 8 MiB, so the spatial axes are +# area-downsampled by the VAE spatial stride before upload. This depends only on +# the stride, not on the H3 shape lattice, which the server still owns. +_VAE_SPATIAL_STRIDE = 16 +_MASK_DECIMALS = 4 + def scalar_mask_to_json(value: float) -> str: if not 0.0 <= value <= 1.0: @@ -13,6 +27,27 @@ def scalar_mask_to_json(value: float) -> str: return str(value) +def video_mask_to_json(mask: torch.Tensor) -> str: + """Serialize a frame-space video mask to JSON for the ``video_noise_mask`` field. + + ``mask`` is a 2D spatial mask or a 3D frame-space mask. Its spatial axes are + area-downsampled by the VAE stride; the server resizes the result to the + latent grid, so the client neither floors the canvas nor maps frames to + latents. + """ + if mask.ndim not in (2, 3): + raise ValueError(f"expected a 2D or 3D mask tensor, got {mask.ndim}D") + frames = mask.unsqueeze(0) if mask.ndim == 2 else mask + size = tuple(max(1, math.ceil(dim / _VAE_SPATIAL_STRIDE)) for dim in frames.shape[1:]) + frames = F.interpolate(frames.unsqueeze(1).float(), size=size, mode="area").squeeze(1) + if mask.ndim == 2: + frames = frames.squeeze(0) + # Round in float64 so tolist() emits short decimals rather than float32 noise. + return json.dumps(frames.double().round(decimals=_MASK_DECIMALS).tolist(), separators=(",", ":")) + + +# The temporal-mask node snaps its preserve boundary to the H3 frame lattice; +# this mirrors the server shape planner. def _align_frame_count(frame_count: int) -> int: if frame_count <= 0: return 1 @@ -20,32 +55,3 @@ def _align_frame_count(frame_count: int) -> int: while current % 17 != 5: current += 1 return current - - -def _video_latent_t(frame_count: int) -> int: - if frame_count <= 5: - return 2 - return ((int(frame_count) - 5) // 17) * 5 + 2 - - -def video_mask_to_grid(mask: torch.Tensor, *, width: int, height: int, num_frames: int) -> torch.Tensor: - tv = _video_latent_t(_align_frame_count(num_frames)) - height = int(height) // 32 * 32 - width = int(width) // 32 * 32 - gh, gw = height // 16, width // 16 - - if mask.ndim == 2: - mask = mask.unsqueeze(0) - elif mask.ndim != 3: - raise ValueError(f"expected a 2D or 3D mask tensor, got {mask.ndim}D") - - grid = F.interpolate(mask.unsqueeze(1).float(), size=(gh, gw), mode="area").squeeze(1) - if grid.shape[0] == 1: - grid = grid.expand(tv, gh, gw) - elif grid.shape[0] != tv: - grid = F.interpolate(grid.unsqueeze(0).unsqueeze(0), size=(tv, gh, gw), mode="nearest").squeeze(0).squeeze(0) - return grid - - -def video_mask_to_grid_json(mask: torch.Tensor, *, width: int, height: int, num_frames: int) -> str: - return json.dumps(video_mask_to_grid(mask, width=width, height=height, num_frames=num_frames).tolist()) diff --git a/apps/ComfyUI-vLLM-Omni/docs/wf05-h3-latent-editing.md b/apps/ComfyUI-vLLM-Omni/docs/wf05-h3-latent-editing.md index 9ed10273f13..75a45d60d0d 100644 --- a/apps/ComfyUI-vLLM-Omni/docs/wf05-h3-latent-editing.md +++ b/apps/ComfyUI-vLLM-Omni/docs/wf05-h3-latent-editing.md @@ -12,6 +12,6 @@ Edit a source video with MiniMax-H3 using spatial or temporal masks. The workflo Open **vLLM-Omni MiniMax-H3 Latent Mask Editing.json** from the example workflows and set the Generate Video URL/model for the case you want to run. 1. Upload your source video in Load Video. -2. Adjust the case's mask. For removal and inpainting, set Mask Canvas to the video dimensions, Mask Rectangle to the region size, and Combine Masks x/y to its position. For continuation and extension, set MiniMax-H3 Temporal Mask's mode and duration; continuation also uses `preserve_fraction`. Match the duration in Generate Video, and use a duration longer than the source for extension. Mask values are `0` to preserve and `1` to regenerate. +2. Adjust the case's mask. For removal and inpainting, set Mask Canvas to the video dimensions, Mask Rectangle to the region size, and Combine Masks x/y to its position, or replace that chain with Load Image (as Mask) for an arbitrary-shape mask. For continuation and extension, set MiniMax-H3 Temporal Mask's mode and duration; continuation also uses `preserve_fraction`. Match the duration in Generate Video, and use a duration longer than the source for extension. Mask values are `0` to preserve and `1` to regenerate. 3. Adjust the prompt to describe the desired content. Set the output dimensions and duration in Generate Video. If editing audio, update the sound description and audio mask (`0` preserves, `1` regenerates the whole clip). 4. Select the case's Mask Preview Save Video node and choose **Execute to selected output nodes** to preview the mask as a red overlay without inference, or select its Result Save Video node to generate the edited video. Running the entire workflow generates all four cases. diff --git a/apps/ComfyUI-vLLM-Omni/example_workflows/vLLM-Omni Latent Mask Editing - Image Mask.json b/apps/ComfyUI-vLLM-Omni/example_workflows/vLLM-Omni Latent Mask Editing - Image Mask.json deleted file mode 100644 index 17be731bff7..00000000000 --- a/apps/ComfyUI-vLLM-Omni/example_workflows/vLLM-Omni Latent Mask Editing - Image Mask.json +++ /dev/null @@ -1,499 +0,0 @@ -{ - "id": "latent-mask-editing-7380", - "revision": 0, - "last_node_id": 7, - "last_link_id": 5, - "nodes": [ - { - "id": 1, - "type": "MarkdownNote", - "pos": [ - 40, - 20 - ], - "size": [ - 420, - 210 - ], - "flags": {}, - "order": 0, - "mode": 0, - "inputs": [], - "outputs": [], - "title": "Latent-mask editing server prerequisite", - "properties": {}, - "widgets_values": [ - "Requires a vLLM-Omni server that declares latent-mask editing support (MiniMax-H3, EDIT-01). The source clip and video/audio noise masks are uploaded to /v1/videos; 0 preserves, 1 regenerates, fractional blends." - ], - "color": "#432", - "bgcolor": "#000" - }, - { - "id": 2, - "type": "LoadVideo", - "pos": [ - 40, - 280 - ], - "size": [ - 420, - 120 - ], - "flags": {}, - "order": 1, - "mode": 0, - "inputs": [ - { - "localized_name": "file", - "name": "file", - "type": "COMBO", - "widget": { - "name": "file" - }, - "link": null - } - ], - "outputs": [ - { - "localized_name": "video", - "name": "video", - "type": "VIDEO", - "links": [ - 1 - ] - } - ], - "properties": { - "Node name for S&R": "LoadVideo" - }, - "widgets_values": [ - "" - ] - }, - { - "id": 3, - "type": "LoadImageMask", - "pos": [ - 40, - 460 - ], - "size": [ - 420, - 160 - ], - "flags": {}, - "order": 2, - "mode": 0, - "inputs": [ - { - "name": "image", - "type": "STRING", - "widget": { - "name": "image" - }, - "link": null - }, - { - "name": "channel", - "type": "STRING", - "widget": { - "name": "channel" - }, - "link": null - } - ], - "outputs": [ - { - "localized_name": "mask", - "name": "mask", - "type": "MASK", - "links": [ - 2 - ] - } - ], - "properties": { - "Node name for S&R": "LoadImageMask" - }, - "widgets_values": [ - "", - "red" - ] - }, - { - "id": 4, - "type": "VLLMOmniLatentMaskEditing", - "pos": [ - 560, - 260 - ], - "size": [ - 420, - 300 - ], - "flags": {}, - "order": 3, - "mode": 0, - "inputs": [ - { - "localized_name": "source_video", - "name": "source_video", - "type": "VIDEO", - "link": 1, - "shape": 7 - }, - { - "localized_name": "source_audio", - "name": "source_audio", - "type": "AUDIO", - "link": null - }, - { - "localized_name": "video_mask", - "name": "video_mask", - "type": "MASK", - "link": 2, - "shape": 7 - }, - { - "localized_name": "audio_mask", - "name": "audio_mask", - "type": "FLOAT", - "widget": { - "name": "audio_mask" - }, - "link": null - } - ], - "outputs": [ - { - "localized_name": "latent_edit", - "name": "latent_edit", - "type": "LATENT_MASK_EDITING", - "links": [ - 3 - ] - } - ], - "properties": { - "Node name for S&R": "VLLMOmniLatentMaskEditing" - }, - "widgets_values": [ - 0.5 - ] - }, - { - "id": 5, - "type": "VLLMOmniMiniMaxH3Params", - "pos": [ - 40, - 680 - ], - "size": [ - 420, - 140 - ], - "flags": {}, - "order": 4, - "mode": 0, - "inputs": [ - { - "localized_name": "audio_flow_shift", - "name": "audio_flow_shift", - "type": "FLOAT", - "widget": { - "name": "audio_flow_shift" - }, - "link": null - }, - { - "localized_name": "flow_shift", - "name": "flow_shift", - "type": "FLOAT", - "widget": { - "name": "flow_shift" - }, - "link": null - } - ], - "outputs": [ - { - "localized_name": "video params", - "name": "video params", - "type": "VIDEO_PARAMS", - "links": [ - 4 - ] - } - ], - "properties": { - "Node name for S&R": "VLLMOmniMiniMaxH3Params" - }, - "widgets_values": [ - 3.0, - 12.0 - ] - }, - { - "id": 6, - "type": "VLLMOmniGenerateVideo", - "pos": [ - 1080, - 200 - ], - "size": [ - 540, - 680 - ], - "flags": {}, - "order": 5, - "mode": 0, - "inputs": [ - { - "localized_name": "frame", - "name": "frame", - "type": "IMAGE", - "link": null, - "shape": 7 - }, - { - "localized_name": "references", - "name": "references", - "type": "VIDEO_REFERENCES", - "link": null, - "shape": 7 - }, - { - "localized_name": "sampling_params", - "name": "sampling_params", - "type": "SAMPLING_PARAMS", - "link": null - }, - { - "localized_name": "lora", - "name": "lora", - "type": "REMOTE_LORA", - "link": null - }, - { - "localized_name": "model_params", - "name": "model_params", - "type": "VIDEO_PARAMS", - "link": 4, - "shape": 7 - }, - { - "localized_name": "fast_h3", - "name": "fast_h3", - "type": "FASTH3_DEPLOYMENT", - "link": null - }, - { - "localized_name": "latent_edit", - "name": "latent_edit", - "type": "LATENT_MASK_EDITING", - "link": 3, - "shape": 7 - }, - { - "localized_name": "url", - "name": "url", - "type": "STRING", - "widget": { - "name": "url" - }, - "link": null - }, - { - "localized_name": "model", - "name": "model", - "type": "STRING", - "widget": { - "name": "model" - }, - "link": null - }, - { - "localized_name": "prompt", - "name": "prompt", - "type": "STRING", - "widget": { - "name": "prompt" - }, - "link": null - }, - { - "localized_name": "negative_prompt", - "name": "negative_prompt", - "type": "STRING", - "widget": { - "name": "negative_prompt" - }, - "link": null - }, - { - "localized_name": "width", - "name": "width", - "type": "INT", - "widget": { - "name": "width" - }, - "link": null - }, - { - "localized_name": "height", - "name": "height", - "type": "INT", - "widget": { - "name": "height" - }, - "link": null - }, - { - "localized_name": "fps", - "name": "fps", - "type": "INT", - "widget": { - "name": "fps" - }, - "link": null - }, - { - "localized_name": "duration", - "name": "duration", - "type": "FLOAT", - "widget": { - "name": "duration" - }, - "link": null - } - ], - "outputs": [ - { - "localized_name": "video", - "name": "video", - "type": "VIDEO", - "links": [ - 5 - ] - } - ], - "properties": { - "Node name for S&R": "VLLMOmniGenerateVideo" - }, - "widgets_values": [ - "http://127.0.0.1:8000/v1", - "MiniMaxAI/MiniMax-H3", - "Partially restyle the clip and soundtrack", - "", - 160, - 120, - 24, - 4.458 - ] - }, - { - "id": 7, - "type": "SaveVideo", - "pos": [ - 1700, - 300 - ], - "size": [ - 420, - 160 - ], - "flags": {}, - "order": 6, - "mode": 0, - "inputs": [ - { - "localized_name": "video", - "name": "video", - "type": "VIDEO", - "link": 5, - "shape": 7 - }, - { - "localized_name": "filename_prefix", - "name": "filename_prefix", - "type": "STRING", - "widget": { - "name": "filename_prefix" - }, - "link": null - }, - { - "localized_name": "format", - "name": "format", - "type": "COMFY_DYNAMICCOMBO_V3", - "widget": { - "name": "format" - }, - "link": null - } - ], - "outputs": [ - { - "localized_name": "video", - "name": "video", - "type": "VIDEO", - "links": [] - } - ], - "properties": { - "Node name for S&R": "SaveVideo" - }, - "widgets_values": [ - "vllm-omni/latent-edit", - "mp4" - ] - } - ], - "links": [ - [ - 1, - 2, - 0, - 4, - 0, - "VIDEO" - ], - [ - 2, - 3, - 0, - 4, - 2, - "MASK" - ], - [ - 3, - 4, - 0, - 6, - 6, - "LATENT_MASK_EDITING" - ], - [ - 4, - 5, - 0, - 6, - 4, - "VIDEO_PARAMS" - ], - [ - 5, - 6, - 0, - 7, - 0, - "VIDEO" - ] - ], - "groups": [], - "config": {}, - "extra": {}, - "version": 0.4 -} diff --git a/apps/ComfyUI-vLLM-Omni/example_workflows/vLLM-Omni Latent Mask Editing - Temporal Mask.json b/apps/ComfyUI-vLLM-Omni/example_workflows/vLLM-Omni Latent Mask Editing - Temporal Mask.json deleted file mode 100644 index b2b3a698085..00000000000 --- a/apps/ComfyUI-vLLM-Omni/example_workflows/vLLM-Omni Latent Mask Editing - Temporal Mask.json +++ /dev/null @@ -1,560 +0,0 @@ -{ - "last_node_id": 9, - "last_link_id": 7, - "nodes": [ - { - "id": 2, - "type": "LoadVideo", - "pos": [ - 40, - 100 - ], - "size": [ - 420, - 120 - ], - "flags": {}, - "order": 1, - "mode": 0, - "inputs": [ - { - "name": "file", - "type": "STRING", - "widget": { - "name": "file" - }, - "link": null - } - ], - "outputs": [ - { - "name": "video", - "type": "VIDEO", - "links": [ - 1 - ] - } - ], - "properties": { - "Node name for S&R": "LoadVideo" - }, - "widgets_values": [ - "" - ] - }, - { - "id": 3, - "type": "SolidMask", - "pos": [ - 40, - 300 - ], - "size": [ - 420, - 160 - ], - "flags": {}, - "order": 2, - "mode": 0, - "inputs": [ - { - "name": "value", - "type": "FLOAT", - "widget": { - "name": "value" - }, - "link": null - }, - { - "name": "width", - "type": "INT", - "widget": { - "name": "width" - }, - "link": null - }, - { - "name": "height", - "type": "INT", - "widget": { - "name": "height" - }, - "link": null - } - ], - "outputs": [ - { - "name": "mask", - "type": "MASK", - "links": [ - 2 - ] - } - ], - "properties": { - "Node name for S&R": "SolidMask" - }, - "widgets_values": [ - 0.0, - 160, - 120 - ] - }, - { - "id": 4, - "type": "SolidMask", - "pos": [ - 40, - 500 - ], - "size": [ - 420, - 160 - ], - "flags": {}, - "order": 3, - "mode": 0, - "inputs": [ - { - "name": "value", - "type": "FLOAT", - "widget": { - "name": "value" - }, - "link": null - }, - { - "name": "width", - "type": "INT", - "widget": { - "name": "width" - }, - "link": null - }, - { - "name": "height", - "type": "INT", - "widget": { - "name": "height" - }, - "link": null - } - ], - "outputs": [ - { - "name": "mask", - "type": "MASK", - "links": [ - 3 - ] - } - ], - "properties": { - "Node name for S&R": "SolidMask" - }, - "widgets_values": [ - 1.0, - 160, - 120 - ] - }, - { - "id": 5, - "type": "BatchMasksNode", - "pos": [ - 250, - 380 - ], - "size": [ - 200, - 100 - ], - "flags": {}, - "order": 4, - "mode": 0, - "inputs": [ - { - "name": "masks.mask0", - "type": "MASK", - "link": 2, - "slot_index": 0 - }, - { - "name": "masks.mask1", - "type": "MASK", - "link": 3, - "slot_index": 1 - } - ], - "outputs": [ - { - "name": "MASK", - "type": "MASK", - "links": [ - 4 - ] - } - ], - "properties": { - "Node name for S&R": "BatchMasksNode" - }, - "widgets_values": [] - }, - { - "id": 6, - "type": "VLLMOmniLatentMaskEditing", - "pos": [ - 450, - 180 - ], - "size": [ - 400, - 120 - ], - "flags": {}, - "order": 5, - "mode": 0, - "inputs": [ - { - "name": "source_video", - "type": "VIDEO", - "link": 1 - }, - { - "name": "source_audio", - "type": "AUDIO", - "link": null - }, - { - "name": "video_mask", - "type": "MASK", - "link": 4 - }, - { - "name": "audio_mask", - "type": "FLOAT", - "widget": { - "name": "audio_mask" - }, - "link": null - } - ], - "outputs": [ - { - "name": "latent_edit", - "type": "LATENT_MASK_EDITING", - "links": [ - 5 - ] - } - ], - "properties": { - "Node name for S&R": "VLLMOmniLatentMaskEditing" - }, - "widgets_values": [ - -1.0 - ] - }, - { - "id": 7, - "type": "VLLMOmniMiniMaxH3Params", - "pos": [ - 40, - 700 - ], - "size": [ - 420, - 160 - ], - "flags": {}, - "order": 6, - "mode": 0, - "inputs": [ - { - "name": "audio_flow_shift", - "type": "FLOAT", - "widget": { - "name": "audio_flow_shift" - }, - "link": null - }, - { - "name": "flow_shift", - "type": "FLOAT", - "widget": { - "name": "flow_shift" - }, - "link": null - } - ], - "outputs": [ - { - "name": "video params", - "type": "VIDEO_PARAMS", - "links": [ - 6 - ] - } - ], - "properties": { - "Node name for S&R": "VLLMOmniMiniMaxH3Params" - }, - "widgets_values": [ - 3.0, - 12.0 - ] - }, - { - "id": 8, - "type": "VLLMOmniGenerateVideo", - "pos": [ - 700, - 180 - ], - "size": [ - 450, - 380 - ], - "flags": {}, - "order": 7, - "mode": 0, - "inputs": [ - { - "name": "frame", - "type": "IMAGE", - "link": null - }, - { - "name": "references", - "type": "VIDEO_REFERENCES", - "link": null - }, - { - "name": "sampling_params", - "type": "SAMPLING_PARAMS", - "link": null - }, - { - "name": "lora", - "type": "LORA", - "link": null - }, - { - "name": "model_params", - "type": "VIDEO_PARAMS", - "link": 6 - }, - { - "name": "fast_h3", - "type": "FAST_H3", - "link": null - }, - { - "name": "latent_edit", - "type": "LATENT_MASK_EDITING", - "link": 5 - }, - { - "name": "url", - "type": "STRING", - "widget": { - "name": "url" - }, - "link": null - }, - { - "name": "model", - "type": "STRING", - "widget": { - "name": "model" - }, - "link": null - }, - { - "name": "prompt", - "type": "STRING", - "widget": { - "name": "prompt" - }, - "link": null - }, - { - "name": "negative_prompt", - "type": "STRING", - "widget": { - "name": "negative_prompt" - }, - "link": null - }, - { - "name": "width", - "type": "INT", - "widget": { - "name": "width" - }, - "link": null - }, - { - "name": "height", - "type": "INT", - "widget": { - "name": "height" - }, - "link": null - }, - { - "name": "fps", - "type": "INT", - "widget": { - "name": "fps" - }, - "link": null - }, - { - "name": "duration", - "type": "FLOAT", - "widget": { - "name": "duration" - }, - "link": null - } - ], - "outputs": [ - { - "name": "video", - "type": "VIDEO", - "links": [ - 7 - ] - } - ], - "properties": { - "Node name for S&R": "VLLMOmniGenerateVideo" - }, - "widgets_values": [ - "http://127.0.0.1:8000/v1", - "MiniMaxAI/MiniMax-H3", - "temporal mask: regenerate the ending", - "", - 160, - 120, - 24, - 4.458 - ] - }, - { - "id": 9, - "type": "SaveVideo", - "pos": [ - 900, - 180 - ], - "size": [ - 420, - 200 - ], - "flags": {}, - "order": 8, - "mode": 0, - "inputs": [ - { - "name": "video", - "type": "VIDEO", - "link": 7 - }, - { - "name": "filename_prefix", - "type": "STRING", - "widget": { - "name": "filename_prefix" - }, - "link": null - }, - { - "name": "format", - "type": "STRING", - "widget": { - "name": "format" - }, - "link": null - } - ], - "outputs": [ - { - "name": "video", - "type": "VIDEO", - "links": [] - } - ], - "properties": { - "Node name for S&R": "SaveVideo" - }, - "widgets_values": [ - "temporal_mask", - "mp4" - ] - } - ], - "links": [ - [ - 1, - 2, - 0, - 6, - 0, - "VIDEO" - ], - [ - 2, - 3, - 0, - 5, - 0, - "MASK" - ], - [ - 3, - 4, - 0, - 5, - 1, - "MASK" - ], - [ - 4, - 5, - 0, - 6, - 2, - "MASK" - ], - [ - 5, - 6, - 0, - 8, - 6, - "LATENT_MASK_EDITING" - ], - [ - 6, - 7, - 0, - 8, - 4, - "VIDEO_PARAMS" - ], - [ - 7, - 8, - 0, - 9, - 0, - "VIDEO" - ] - ], - "groups": [], - "config": {}, - "extra": {}, - "version": 0.4 -} diff --git a/apps/ComfyUI-vLLM-Omni/example_workflows/vLLM-Omni Latent Mask Editing.json b/apps/ComfyUI-vLLM-Omni/example_workflows/vLLM-Omni Latent Mask Editing.json deleted file mode 100644 index 4192ef4c5fc..00000000000 --- a/apps/ComfyUI-vLLM-Omni/example_workflows/vLLM-Omni Latent Mask Editing.json +++ /dev/null @@ -1,511 +0,0 @@ -{ - "id": "latent-mask-editing-7380", - "revision": 0, - "last_node_id": 7, - "last_link_id": 5, - "nodes": [ - { - "id": 1, - "type": "MarkdownNote", - "pos": [ - 40, - 20 - ], - "size": [ - 420, - 210 - ], - "flags": {}, - "order": 0, - "mode": 0, - "inputs": [], - "outputs": [], - "title": "Latent-mask editing server prerequisite", - "properties": {}, - "widgets_values": [ - "Requires a vLLM-Omni server that declares latent-mask editing support (MiniMax-H3, EDIT-01). The source clip and video/audio noise masks are uploaded to /v1/videos; 0 preserves, 1 regenerates, fractional blends." - ], - "color": "#432", - "bgcolor": "#000" - }, - { - "id": 2, - "type": "LoadVideo", - "pos": [ - 40, - 280 - ], - "size": [ - 420, - 120 - ], - "flags": {}, - "order": 1, - "mode": 0, - "inputs": [ - { - "localized_name": "file", - "name": "file", - "type": "COMBO", - "widget": { - "name": "file" - }, - "link": null - } - ], - "outputs": [ - { - "localized_name": "video", - "name": "video", - "type": "VIDEO", - "links": [ - 1 - ] - } - ], - "properties": { - "Node name for S&R": "LoadVideo" - }, - "widgets_values": [ - "" - ] - }, - { - "id": 3, - "type": "SolidMask", - "pos": [ - 40, - 460 - ], - "size": [ - 420, - 160 - ], - "flags": {}, - "order": 2, - "mode": 0, - "inputs": [ - { - "localized_name": "value", - "name": "value", - "type": "FLOAT", - "widget": { - "name": "value" - }, - "link": null - }, - { - "localized_name": "width", - "name": "width", - "type": "INT", - "widget": { - "name": "width" - }, - "link": null - }, - { - "localized_name": "height", - "name": "height", - "type": "INT", - "widget": { - "name": "height" - }, - "link": null - } - ], - "outputs": [ - { - "localized_name": "mask", - "name": "mask", - "type": "MASK", - "links": [ - 2 - ] - } - ], - "properties": { - "Node name for S&R": "SolidMask" - }, - "widgets_values": [ - 1.0, - 160, - 120 - ] - }, - { - "id": 4, - "type": "VLLMOmniLatentMaskEditing", - "pos": [ - 560, - 260 - ], - "size": [ - 420, - 300 - ], - "flags": {}, - "order": 3, - "mode": 0, - "inputs": [ - { - "localized_name": "source_video", - "name": "source_video", - "type": "VIDEO", - "link": 1, - "shape": 7 - }, - { - "localized_name": "source_audio", - "name": "source_audio", - "type": "AUDIO", - "link": null - }, - { - "localized_name": "video_mask", - "name": "video_mask", - "type": "MASK", - "link": 2, - "shape": 7 - }, - { - "localized_name": "audio_mask", - "name": "audio_mask", - "type": "FLOAT", - "widget": { - "name": "audio_mask" - }, - "link": null - } - ], - "outputs": [ - { - "localized_name": "latent_edit", - "name": "latent_edit", - "type": "LATENT_MASK_EDITING", - "links": [ - 3 - ] - } - ], - "properties": { - "Node name for S&R": "VLLMOmniLatentMaskEditing" - }, - "widgets_values": [ - 0.5 - ] - }, - { - "id": 5, - "type": "VLLMOmniMiniMaxH3Params", - "pos": [ - 40, - 680 - ], - "size": [ - 420, - 140 - ], - "flags": {}, - "order": 4, - "mode": 0, - "inputs": [ - { - "localized_name": "audio_flow_shift", - "name": "audio_flow_shift", - "type": "FLOAT", - "widget": { - "name": "audio_flow_shift" - }, - "link": null - }, - { - "localized_name": "flow_shift", - "name": "flow_shift", - "type": "FLOAT", - "widget": { - "name": "flow_shift" - }, - "link": null - } - ], - "outputs": [ - { - "localized_name": "video params", - "name": "video params", - "type": "VIDEO_PARAMS", - "links": [ - 4 - ] - } - ], - "properties": { - "Node name for S&R": "VLLMOmniMiniMaxH3Params" - }, - "widgets_values": [ - 3.0, - 12.0 - ] - }, - { - "id": 6, - "type": "VLLMOmniGenerateVideo", - "pos": [ - 1080, - 200 - ], - "size": [ - 540, - 680 - ], - "flags": {}, - "order": 5, - "mode": 0, - "inputs": [ - { - "localized_name": "frame", - "name": "frame", - "type": "IMAGE", - "link": null, - "shape": 7 - }, - { - "localized_name": "references", - "name": "references", - "type": "VIDEO_REFERENCES", - "link": null, - "shape": 7 - }, - { - "localized_name": "sampling_params", - "name": "sampling_params", - "type": "SAMPLING_PARAMS", - "link": null - }, - { - "localized_name": "lora", - "name": "lora", - "type": "REMOTE_LORA", - "link": null - }, - { - "localized_name": "model_params", - "name": "model_params", - "type": "VIDEO_PARAMS", - "link": 4, - "shape": 7 - }, - { - "localized_name": "fast_h3", - "name": "fast_h3", - "type": "FASTH3_DEPLOYMENT", - "link": null - }, - { - "localized_name": "latent_edit", - "name": "latent_edit", - "type": "LATENT_MASK_EDITING", - "link": 3, - "shape": 7 - }, - { - "localized_name": "url", - "name": "url", - "type": "STRING", - "widget": { - "name": "url" - }, - "link": null - }, - { - "localized_name": "model", - "name": "model", - "type": "STRING", - "widget": { - "name": "model" - }, - "link": null - }, - { - "localized_name": "prompt", - "name": "prompt", - "type": "STRING", - "widget": { - "name": "prompt" - }, - "link": null - }, - { - "localized_name": "negative_prompt", - "name": "negative_prompt", - "type": "STRING", - "widget": { - "name": "negative_prompt" - }, - "link": null - }, - { - "localized_name": "width", - "name": "width", - "type": "INT", - "widget": { - "name": "width" - }, - "link": null - }, - { - "localized_name": "height", - "name": "height", - "type": "INT", - "widget": { - "name": "height" - }, - "link": null - }, - { - "localized_name": "fps", - "name": "fps", - "type": "INT", - "widget": { - "name": "fps" - }, - "link": null - }, - { - "localized_name": "duration", - "name": "duration", - "type": "FLOAT", - "widget": { - "name": "duration" - }, - "link": null - } - ], - "outputs": [ - { - "localized_name": "video", - "name": "video", - "type": "VIDEO", - "links": [ - 5 - ] - } - ], - "properties": { - "Node name for S&R": "VLLMOmniGenerateVideo" - }, - "widgets_values": [ - "http://127.0.0.1:8000/v1", - "MiniMaxAI/MiniMax-H3", - "Partially restyle the clip and soundtrack", - "", - 160, - 120, - 24, - 4.458 - ] - }, - { - "id": 7, - "type": "SaveVideo", - "pos": [ - 1700, - 300 - ], - "size": [ - 420, - 160 - ], - "flags": {}, - "order": 6, - "mode": 0, - "inputs": [ - { - "localized_name": "video", - "name": "video", - "type": "VIDEO", - "link": 5, - "shape": 7 - }, - { - "localized_name": "filename_prefix", - "name": "filename_prefix", - "type": "STRING", - "widget": { - "name": "filename_prefix" - }, - "link": null - }, - { - "localized_name": "format", - "name": "format", - "type": "COMFY_DYNAMICCOMBO_V3", - "widget": { - "name": "format" - }, - "link": null - } - ], - "outputs": [ - { - "localized_name": "video", - "name": "video", - "type": "VIDEO", - "links": [] - } - ], - "properties": { - "Node name for S&R": "SaveVideo" - }, - "widgets_values": [ - "vllm-omni/latent-edit", - "mp4" - ] - } - ], - "links": [ - [ - 1, - 2, - 0, - 4, - 0, - "VIDEO" - ], - [ - 2, - 3, - 0, - 4, - 2, - "MASK" - ], - [ - 3, - 4, - 0, - 6, - 6, - "LATENT_MASK_EDITING" - ], - [ - 4, - 5, - 0, - 6, - 4, - "VIDEO_PARAMS" - ], - [ - 5, - 6, - 0, - 7, - 0, - "VIDEO" - ] - ], - "groups": [], - "config": {}, - "extra": {}, - "version": 0.4 -} diff --git a/apps/ComfyUI-vLLM-Omni/example_workflows/vLLM-Omni MiniMax-H3 Latent Mask Editing.json b/apps/ComfyUI-vLLM-Omni/example_workflows/vLLM-Omni MiniMax-H3 Latent Mask Editing.json index fc434173093..8946f30a171 100644 --- a/apps/ComfyUI-vLLM-Omni/example_workflows/vLLM-Omni MiniMax-H3 Latent Mask Editing.json +++ b/apps/ComfyUI-vLLM-Omni/example_workflows/vLLM-Omni MiniMax-H3 Latent Mask Editing.json @@ -2293,10 +2293,10 @@ "title": "Usage Guide", "properties": {}, "widgets_values": [ - "Use a compatible MiniMax-H3 latent-edit server. Select a source video and set the server URL/model in each Generate Video node.\n\nSelect one Save Preview or Save Video output, then use Execute to selected output nodes. Running the whole workflow submits all four cases.\n\nMask: 0 preserves source latents; 1 regenerates. Audio mask applies to the whole clip, not individual sounds. Preview overlays are silent and do not run inference." + "Use a compatible MiniMax-H3 latent-edit server. Select a source video and set the server URL/model in each Generate Video node.\n\nSelect one Save Preview or Save Video output, then use Execute to selected output nodes. Running the whole workflow submits all four cases.\n\nMask: 0 preserves source latents; 1 regenerates. For an arbitrary-shape spatial mask, replace a case's Mask Rectangle chain with Load Image (as Mask). Audio mask applies to the whole clip, not individual sounds. Preview overlays are silent and do not run inference." ], "widgets_values_named": { - "text": "Use a compatible MiniMax-H3 latent-edit server. Select a source video and set the server URL/model in each Generate Video node.\n\nSelect one Save Preview or Save Video output, then use Execute to selected output nodes. Running the whole workflow submits all four cases.\n\nMask: 0 preserves source latents; 1 regenerates. Audio mask applies to the whole clip, not individual sounds. Preview overlays are silent and do not run inference." + "text": "Use a compatible MiniMax-H3 latent-edit server. Select a source video and set the server URL/model in each Generate Video node.\n\nSelect one Save Preview or Save Video output, then use Execute to selected output nodes. Running the whole workflow submits all four cases.\n\nMask: 0 preserves source latents; 1 regenerates. For an arbitrary-shape spatial mask, replace a case's Mask Rectangle chain with Load Image (as Mask). Audio mask applies to the whole clip, not individual sounds. Preview overlays are silent and do not run inference." }, "color": "#432", "bgcolor": "#000" @@ -2823,7 +2823,7 @@ "Node name for S&R": "VLLMOmniGenerateVideo" }, "widgets_values": [ - "http://127.0.0.1:8001/v1", + "http://127.0.0.1:8000/v1", "MiniMaxAI/MiniMax-H3", "integrated_multimodal_description: The masked area is a flat, seamless continuation of the surrounding wall, matching its color, texture, and lighting. The wall surface remains consistent throughout the video.\n\noverall_soundscape: \n\nnon_diegetic_music: ", "", @@ -2833,7 +2833,7 @@ 5 ], "widgets_values_named": { - "url": "http://127.0.0.1:8001/v1", + "url": "http://127.0.0.1:8000/v1", "model": "MiniMaxAI/MiniMax-H3", "prompt": "integrated_multimodal_description: The masked area is a flat, seamless continuation of the surrounding wall, matching its color, texture, and lighting. The wall surface remains consistent throughout the video.\n\noverall_soundscape: \n\nnon_diegetic_music: ", "negative_prompt": "", @@ -3677,7 +3677,7 @@ "Node name for S&R": "VLLMOmniGenerateVideo" }, "widgets_values": [ - "http://127.0.0.1:8001/v1", + "http://127.0.0.1:8000/v1", "MiniMaxAI/MiniMax-H3", "integrated_multimodal_description: A small orange cat rests curled up in the masked area, gently swaying its tail from side to side. Its appearance matches the surrounding anime style, lighting, and shadows.\n\noverall_soundscape:\n\nnon_diegetic_music: ", "", @@ -3687,7 +3687,7 @@ 5 ], "widgets_values_named": { - "url": "http://127.0.0.1:8001/v1", + "url": "http://127.0.0.1:8000/v1", "model": "MiniMaxAI/MiniMax-H3", "prompt": "integrated_multimodal_description: A small orange cat rests curled up in the masked area, gently swaying its tail from side to side. Its appearance matches the surrounding anime style, lighting, and shadows.\n\noverall_soundscape:\n\nnon_diegetic_music: ", "negative_prompt": "", @@ -3916,7 +3916,7 @@ "Node name for S&R": "VLLMOmniGenerateVideo" }, "widgets_values": [ - "http://127.0.0.1:8001/v1", + "http://127.0.0.1:8000/v1", "MiniMaxAI/MiniMax-H3", "integrated_multimodal_description: The sky turns completely overcast, with the sun hidden and no direct sunlight, as torrential rain pours over the school entrance in dense visible streaks and splashes on the wet pavement. Distant lightning flashes in the sky behind the school buildings.\n\noverall_soundscape: Loud, steady rainfall, splashing water, and gusts of wind.\n\nnon_diegetic_music: None.", "", @@ -3926,7 +3926,7 @@ 5 ], "widgets_values_named": { - "url": "http://127.0.0.1:8001/v1", + "url": "http://127.0.0.1:8000/v1", "model": "MiniMaxAI/MiniMax-H3", "prompt": "integrated_multimodal_description: The sky turns completely overcast, with the sun hidden and no direct sunlight, as torrential rain pours over the school entrance in dense visible streaks and splashes on the wet pavement. Distant lightning flashes in the sky behind the school buildings.\n\noverall_soundscape: Loud, steady rainfall, splashing water, and gusts of wind.\n\nnon_diegetic_music: None.", "negative_prompt": "", @@ -4363,7 +4363,7 @@ "Node name for S&R": "VLLMOmniGenerateVideo" }, "widgets_values": [ - "http://127.0.0.1:8001/v1", + "http://127.0.0.1:8000/v1", "MiniMaxAI/MiniMax-H3", "integrated_multimodal_description: Cut to the same school on a summer night beneath a deep blue sky dotted with stars. A firework rises from behind the school buildings, leaving a glowing trail; the camera slowly tilts upward to follow it into the sky, ending as it bursts into a large bloom of colorful sparks.\n\noverall_soundscape: Soft summer insect sounds, followed by the firework rising with a whistle and bursting with a distant boom and crackling sparks.\n\nnon_diegetic_music: N/A", "", @@ -4373,7 +4373,7 @@ 10 ], "widgets_values_named": { - "url": "http://127.0.0.1:8001/v1", + "url": "http://127.0.0.1:8000/v1", "model": "MiniMaxAI/MiniMax-H3", "prompt": "integrated_multimodal_description: Cut to the same school on a summer night beneath a deep blue sky dotted with stars. A firework rises from behind the school buildings, leaving a glowing trail; the camera slowly tilts upward to follow it into the sky, ending as it bursts into a large bloom of colorful sparks.\n\noverall_soundscape: Soft summer insect sounds, followed by the firework rising with a whistle and bursting with a distant boom and crackling sparks.\n\nnon_diegetic_music: N/A", "negative_prompt": "", diff --git a/tests/e2e/features/comfyui/mock_videos_server.py b/tests/e2e/features/comfyui/mock_videos_server.py deleted file mode 100644 index c2ddfbb5e3d..00000000000 --- a/tests/e2e/features/comfyui/mock_videos_server.py +++ /dev/null @@ -1,115 +0,0 @@ -# SPDX-License-Identifier: Apache-2.0 -# SPDX-FileCopyrightText: Copyright contributors to the vLLM-Omni project - -"""Minimal mock of vLLM-Omni's ``/v1/videos`` API for e2e tests. - -Records every received multipart field to ``MOCK_STATE_FILE`` (JSON) so a test -can assert exactly what the client serialized, and returns a synthetic MP4. -""" - -import asyncio -import json -import os -import subprocess -import uuid - -from fastapi import FastAPI, Request -from fastapi.responses import Response -from starlette.datastructures import UploadFile - -app = FastAPI() -_JOBS: dict[str, dict] = {} -_STATE_FILE = os.environ.get("MOCK_STATE_FILE") - - -def _record(fields: dict) -> None: - if _STATE_FILE: - with open(_STATE_FILE, "w") as f: - json.dump(fields, f) - - -def _synthetic_mp4() -> bytes: - return subprocess.run( - [ - "ffmpeg", - "-hide_banner", - "-loglevel", - "error", - "-y", - "-f", - "lavfi", - "-i", - "testsrc2=size=320x240:rate=24:duration=1", - "-f", - "lavfi", - "-i", - "sine=frequency=440:sample_rate=32000:duration=1", - "-map", - "0:v", - "-map", - "1:a", - "-shortest", - "-c:v", - "libx264", - "-pix_fmt", - "yuv420p", - "-c:a", - "aac", - "-ar", - "32000", - "-ac", - "2", - "-movflags", - "frag_keyframe+empty_moov", - "-f", - "mp4", - "pipe:1", - ], - check=True, - capture_output=True, - ).stdout - - -@app.post("/v1/videos") -async def create_video(request: Request): - form = await request.form() - fields = {} - for key, value in form.items(): - if isinstance(value, UploadFile): - data = await value.read() - fields[key] = { - "file": True, - "size": len(data), - "filename": value.filename, - "content_type": value.content_type, - } - if value.content_type == "application/json": - fields[key]["json"] = json.loads(data) - else: - fields[key] = str(value) - _record(fields) - - job_id = str(uuid.uuid4()) - _JOBS[job_id] = {"status": "queued"} - return {"id": job_id, "status": "queued"} - - -@app.get("/v1/videos/{job_id}") -async def get_video(job_id: str): - if job_id not in _JOBS: - return {"status": "failed", "error": "unknown job"} - return {"id": job_id, "status": "completed"} - - -@app.get("/v1/videos/{job_id}/content") -async def get_video_content(job_id: str): - if job_id not in _JOBS: - return Response(status_code=404) - mp4 = await asyncio.to_thread(_synthetic_mp4) - return Response(content=mp4, media_type="video/mp4") - - -@app.delete("/v1/videos/{job_id}") -async def delete_video(job_id: str): - _JOBS.pop(job_id, None) - return {"deleted": True} diff --git a/tests/e2e/features/comfyui/test_comfyui_integration.py b/tests/e2e/features/comfyui/test_comfyui_integration.py index 113879951f1..a202a48e2ae 100644 --- a/tests/e2e/features/comfyui/test_comfyui_integration.py +++ b/tests/e2e/features/comfyui/test_comfyui_integration.py @@ -30,6 +30,7 @@ VLLMOmniGenerateImage, VLLMOmniGenerateMusic, VLLMOmniGenerateVideo, + VLLMOmniLatentMaskEditing, VLLMOmniTTS, VLLMOmniUnderstanding, VLLMOmniVideoReferences, @@ -1344,3 +1345,52 @@ async def test_music_generation_node_minimax_music3(api_server: str, sampling_ca assert len(result) == 1 assert result[0]["sample_rate"] == 24000 assert result[0]["waveform"].shape == (1, 1, 24000) + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + "server_case", + [ + pytest.param( + ServerCase( + served_model="MiniMaxAI/MiniMax-H3", + stage_list=["diffusion"], + stage_configs=[H3_STAGE_CONFIG], + outputs=[_build_diffusion_video_output()], + ), + id="minimax-h3", + ), + ], + indirect=True, +) +@pytest.mark.parametrize( + "sampling_case", + [pytest.param(SamplingCase(kind=SamplingKind.VIDEO_NONE, sampling_params=None), id="latent-mask")], + indirect=True, +) +async def test_video_generation_node_minimax_h3_latent_mask_editing(api_server: str, sampling_case: SamplingCase): + """Latent-mask editing: the raw video/audio masks must be accepted by the real /v1/videos form parser.""" + mask_node = VLLMOmniLatentMaskEditing() + (latent_edit,) = mask_node.get_latent_edit( + source_video=VideoInput(b"mock-source-video"), + video_mask=torch.zeros((VIDEO_HEIGHT, VIDEO_WIDTH), dtype=torch.float32), + audio_mask=0.5, + ) + + node = VLLMOmniGenerateVideo() + result = await node.generate( + url=api_server, + model="MiniMaxAI/MiniMax-H3", + prompt="Restyle the clip.", + negative_prompt="", + width=VIDEO_WIDTH, + height=VIDEO_HEIGHT, + fps=VIDEO_FPS, + duration=VIDEO_DURATION, + model_params=H3_MODEL_PARAMS, + latent_edit=latent_edit, + ) + + assert isinstance(result, tuple) + assert len(result) == 1 + assert isinstance(result[0], VideoInput) diff --git a/tests/e2e/features/comfyui/test_latent_mask.py b/tests/e2e/features/comfyui/test_latent_mask.py index e208fd5b4f9..4ef81878fea 100644 --- a/tests/e2e/features/comfyui/test_latent_mask.py +++ b/tests/e2e/features/comfyui/test_latent_mask.py @@ -1,13 +1,14 @@ # SPDX-License-Identifier: Apache-2.0 # SPDX-FileCopyrightText: Copyright contributors to the vLLM-Omni project -"""Unit tests for latent-mask serialization helpers.""" +"""Unit tests for latent-mask serialization helpers and the temporal-mask node.""" import json import pytest import torch -from comfyui_vllm_omni.utils.latent_mask import scalar_mask_to_json, video_mask_to_grid_json +from comfyui_vllm_omni.nodes import VLLMOmniMiniMaxH3TemporalMask +from comfyui_vllm_omni.utils.latent_mask import scalar_mask_to_json, video_mask_to_json pytestmark = [pytest.mark.core_model, pytest.mark.cpu] @@ -21,60 +22,60 @@ def test_scalar_mask_to_json(): scalar_mask_to_json(-0.1) -def test_video_mask_grid_shape(): - mask = torch.zeros(256, 448) - grid = json.loads(video_mask_to_grid_json(mask, width=448, height=256, num_frames=107)) - assert len(grid) == 2 + 5 * ((107 - 5) // 17) - assert len(grid[0]) == 256 // 16 - assert len(grid[0][0]) == 448 // 16 +def test_video_mask_to_json_downsamples_spatial_mask(): + # A 2D mask keeps its rank; only the spatial axes shrink by the VAE stride. + grid = json.loads(video_mask_to_json(torch.full((120, 160), 0.5))) + assert len(grid) == 8 # ceil(120 / 16) + assert len(grid[0]) == 10 + assert grid[0][0] == 0.5 -def test_video_mask_grid_shape_non_aligned_frames(): - # 96 snaps up to 107 (17n+5), so Tv == 32, matching the server. - grid = json.loads(video_mask_to_grid_json(torch.zeros(256, 448), width=448, height=256, num_frames=96)) - assert len(grid) == 32 +def test_video_mask_to_json_keeps_frame_axis(): + # A 3D mask keeps one slice per frame; the server maps frames to latents. + grid = json.loads(video_mask_to_json(torch.full((22, 120, 160), 0.5))) + assert len(grid) == 22 + assert len(grid[0]) == 8 + assert len(grid[0][0]) == 10 -def test_video_mask_grid_shape_short_clip(): - # num_frames <= 5 -> Tv == 2. - grid = json.loads(video_mask_to_grid_json(torch.zeros(64, 64), width=64, height=64, num_frames=1)) - assert len(grid) == 2 +def test_video_mask_to_json_area_averages_and_rounds(): + mask = torch.zeros(16, 32) + mask[:, 16:] = 1.0 + mask[0, :16] = 1.0 / 3.0 + assert json.loads(video_mask_to_json(mask)) == [[0.0208, 1.0]] -def test_video_mask_grid_shape_unaligned_width(): - # width=500 floors to 480 (multiple of 32) -> gw == 30, not 31. - grid = json.loads(video_mask_to_grid_json(torch.zeros(64, 64), width=500, height=256, num_frames=22)) - assert len(grid[0][0]) == 30 +def test_video_mask_to_json_fits_upload_limit(): + # A soft full-resolution 1344x768 mask serialized raw would exceed 8 MiB. + payload = video_mask_to_json(torch.rand(768, 1344)) + assert len(payload.encode()) < 8 * 1024 * 1024 // 100 -def test_video_mask_3d_input(): - grid = json.loads(video_mask_to_grid_json(torch.ones(1, 256, 448), width=448, height=256, num_frames=107)) - assert grid[0][0][0] == 1.0 - - -def test_video_mask_temporal(): - # 3D mask: first half 0 (preserve), second half 1 (regenerate). - mask = torch.zeros(7, 64, 64) - mask[4:] = 1.0 - grid = json.loads(video_mask_to_grid_json(mask, width=64, height=64, num_frames=22)) - assert len(grid) == 7 - assert all(v == 0.0 for row in grid[0] for v in row) - assert all(v == 1.0 for row in grid[-1] for v in row) - - -def test_video_mask_temporal_resize(): - # 2 temporal slices resize to tv == 7. - mask = torch.zeros(2, 64, 64) - mask[1] = 1.0 - grid = json.loads(video_mask_to_grid_json(mask, width=64, height=64, num_frames=22)) - assert len(grid) == 7 - - -def test_video_mask_uniform_preserved(): - grid = json.loads(video_mask_to_grid_json(torch.full((64, 64), 0.5), width=64, height=64, num_frames=107)) - assert all(v == 0.5 for row in grid for cell in row for v in cell) - - -def test_video_mask_values_in_range(): - grid = json.loads(video_mask_to_grid_json(torch.rand(128, 128), width=128, height=128, num_frames=22)) - assert all(0.0 <= v <= 1.0 for row in grid for cell in row for v in cell) +def test_video_mask_to_json_rejects_wrong_rank(): + with pytest.raises(ValueError): + video_mask_to_json(torch.zeros(4)) + with pytest.raises(ValueError): + video_mask_to_json(torch.zeros(1, 1, 1, 1)) + + +@pytest.mark.parametrize( + ("mode", "duration", "preserve_fraction", "frames", "prefix"), + [ + # 5 s source extended to 10 s: 240 frames align to 243; the 120 source + # frames snap down to a 107-frame prefix (32 latents on the server). + ("extension", 10.0, 0.5, 243, 107), + # 5 s output keeps 20% of the source: 24 frames snap down to 22. + ("continuation", 5.0, 0.2, 124, 22), + ], +) +def test_temporal_mask_is_frame_space(mode, duration, preserve_fraction, frames, prefix): + images = torch.zeros(120, 8, 8, 3) + mask, preview_fps, preview_images, preview_mask = VLLMOmniMiniMaxH3TemporalMask().build( + images, 24.0, duration, mode, preserve_fraction + ) + assert mask.shape == (frames, 1, 1) + assert bool((mask[:prefix] == 0).all()) + assert bool((mask[prefix:] == 1).all()) + assert preview_fps == 24.0 + assert preview_images.shape[0] == frames + assert torch.equal(preview_mask, mask) diff --git a/tests/e2e/features/comfyui/test_latent_mask_editing.py b/tests/e2e/features/comfyui/test_latent_mask_editing.py index 9162974b667..b2f8519dbeb 100644 --- a/tests/e2e/features/comfyui/test_latent_mask_editing.py +++ b/tests/e2e/features/comfyui/test_latent_mask_editing.py @@ -3,97 +3,80 @@ """End-to-end test for latent-mask editing serialization. -Exercises ``VLLMOmniClient.generate_video`` with a ``latent_edit`` payload -against a mock ``/v1/videos`` server and asserts the multipart fields the -client produced. Runs on CPU: the ComfyUI ``comfy_api`` / ``comfy_extras`` -modules are mocked by this directory's ``conftest.py``. +Exercises ``VLLMOmniClient.generate_video`` with a ``latent_edit`` payload and +asserts the exact multipart fields the client produces — in particular that the +masks are uploaded as JSON *file* parts (the server's ``_parse_video_form`` +declares ``video_noise_mask``/``audio_noise_mask`` as ``UploadFile``, so a plain +string field would be rejected with a 422). Runs on CPU: ``comfy_api`` / +``comfy_extras`` are mocked by this directory's ``conftest.py``. """ -import asyncio import json -import os -import socket -import subprocess -import sys -import time +from unittest.mock import AsyncMock import pytest import torch from comfy_api.input import VideoInput -from comfyui_vllm_omni.utils.api_client import VLLMOmniClient +from comfyui_vllm_omni.utils import api_client pytestmark = [pytest.mark.core_model, pytest.mark.cpu] -_TESTS_DIR = os.path.dirname(os.path.abspath(__file__)) - - -def _free_port() -> int: - with socket.socket() as s: - s.bind(("127.0.0.1", 0)) - return s.getsockname()[1] - @pytest.fixture -def mock_server(tmp_path): - port = _free_port() - state_file = tmp_path / "state.json" - env = dict(os.environ, MOCK_STATE_FILE=str(state_file)) - proc = subprocess.Popen( - [sys.executable, "-m", "uvicorn", "mock_videos_server:app", "--host", "127.0.0.1", "--port", str(port)], - cwd=_TESTS_DIR, - env=env, - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, +def api_calls(monkeypatch): + calls = AsyncMock(return_value={"id": "test-video", "status": "completed"}) + monkeypatch.setattr(api_client, "url_json", calls) + monkeypatch.setattr(api_client, "url_bytes", AsyncMock(return_value=b"generated-video")) + monkeypatch.setattr(api_client, "bytes_to_video", lambda data: data) + return calls + + +def _fields_by_name(calls): + fields = calls.call_args_list[0].kwargs["data"]._fields + return {options["name"]: (options, headers, value) for options, headers, value in fields} + + +async def _generate(**kwargs): + return await api_client.VLLMOmniClient("http://localhost/v1").generate_video( + model="MiniMaxAI/MiniMax-H3", + prompt="restyle the clip", + width=160, + height=120, + num_frames=22, + fps=24, + **kwargs, ) - try: - deadline = time.time() + 30 - while time.time() < deadline: - try: - with socket.create_connection(("127.0.0.1", port), timeout=1): - break - except OSError: - time.sleep(0.2) - yield f"http://127.0.0.1:{port}/v1", state_file - finally: - proc.terminate() - proc.wait(timeout=10) - - -def test_latent_edit_serialization(mock_server): - base_url, state_file = mock_server - - # A non-trivial video mask requires a source video. The mocked VideoInput - # only needs to provide ``save_to`` for the client's multipart upload. - source_video = VideoInput(b"mock_source_video") - mask = torch.zeros(1, 120, 160) - latent_edit = {"source_video": source_video, "video_mask": mask, "audio_mask": 0.5} - - async def run(): - client = VLLMOmniClient(base_url) - return await client.generate_video( - model="MiniMaxAI/MiniMax-H3", - prompt="restyle the clip", - width=160, - height=120, - num_frames=22, - fps=24, - latent_edit=latent_edit, - ) - - out = asyncio.run(run()) - assert out is not None - - with open(state_file) as f: - fields = json.load(f) - - assert fields["source_video"]["file"] is True - assert fields["source_video"]["content_type"] == "video/mp4" - for name in ("video_noise_mask", "audio_noise_mask"): - assert fields[name]["file"] is True - assert fields[name]["content_type"] == "application/json" - assert fields["audio_noise_mask"]["json"] == 0.5 - - video_mask = fields["video_noise_mask"]["json"] - assert len(video_mask) == 7 # num_frames=22 aligns to 22 (17n+5) -> Tv = 7 - assert len(video_mask[0]) == 6 # height 120 floors to 96 (multiple of 32) -> gh 6 - assert len(video_mask[0][0]) == 10 # width 160 -> gw 10 + + +async def test_latent_edit_serialization(api_calls): + latent_edit = { + "source_video": VideoInput(b"mock_source_video"), + "video_mask": torch.zeros(1, 120, 160), + "audio_mask": 0.5, + } + + out = await _generate(latent_edit=latent_edit) + assert out == b"generated-video" + + fields = _fields_by_name(api_calls) + + source = fields["source_video"] + assert source[0]["filename"] == "source.mp4" + assert source[1]["Content-Type"] == "video/mp4" + + # video_noise_mask must be a JSON file part, not a string field. The mask + # keeps its frame axis and is downsampled by the VAE stride (the server + # resolves the latent grid). + video = fields["video_noise_mask"] + assert video[0]["filename"] == "video-mask.json" + assert video[1]["Content-Type"] == "application/json" + grid = json.loads(video[2].decode()) + assert len(grid) == 1 # one temporal slice + assert len(grid[0]) == 8 # ceil(120 / 16) + assert len(grid[0][0]) == 10 # 160 / 16 + + # A scalar audio_mask is sent as a JSON file part carrying the scalar value. + audio = fields["audio_noise_mask"] + assert audio[0]["filename"] == "audio-mask.json" + assert audio[1]["Content-Type"] == "application/json" + assert audio[2].decode() == "0.5" diff --git a/tests/e2e/features/comfyui/test_latent_mask_workflows.py b/tests/e2e/features/comfyui/test_latent_mask_workflows.py new file mode 100644 index 00000000000..e337963f7c0 --- /dev/null +++ b/tests/e2e/features/comfyui/test_latent_mask_workflows.py @@ -0,0 +1,97 @@ +# SPDX-License-Identifier: Apache-2.0 +# SPDX-FileCopyrightText: Copyright contributors to the vLLM-Omni project + +import json +from graphlib import TopologicalSorter +from pathlib import Path + +import pytest +from comfyui_vllm_omni import nodes as omni_nodes + +pytestmark = [pytest.mark.core_model, pytest.mark.cpu] + +WORKFLOW = ( + Path(__file__).resolve().parents[4] + / "apps/ComfyUI-vLLM-Omni/example_workflows/vLLM-Omni MiniMax-H3 Latent Mask Editing.json" +) + + +@pytest.fixture +def workflow(): + return json.loads(WORKFLOW.read_text()) + + +def _source_of(workflow: dict, node: dict, input_name: str) -> dict: + nodes = {n["id"]: n for n in workflow["nodes"]} + links = {link[0]: link for link in workflow["links"]} + link_id = next(i["link"] for i in node["inputs"] if i["name"] == input_name) + assert link_id is not None, f"{node['type']}.{input_name} is not connected" + return nodes[links[link_id][1]] + + +def _cases(workflow: dict) -> dict[str, dict]: + return { + node["title"].removeprefix("Generate Video(").removesuffix(")"): node + for node in workflow["nodes"] + if node["type"] == "VLLMOmniGenerateVideo" + } + + +def test_latent_mask_workflow_connections(workflow): + nodes = {node["id"]: node for node in workflow["nodes"]} + links = {link[0]: link for link in workflow["links"]} + assert len(nodes) == len(workflow["nodes"]) + assert len(links) == len(workflow["links"]) + graph: dict[int, set[int]] = {node_id: set() for node_id in nodes} + for link_id, source, output_slot, target, input_slot, kind in links.values(): + output = nodes[source]["outputs"][output_slot] + input_ = nodes[target]["inputs"][input_slot] + assert output["type"] == input_["type"] == kind + assert link_id in output["links"] + assert input_["link"] == link_id + graph[target].add(source) + assert len(tuple(TopologicalSorter(graph).static_order())) == len(nodes) + + +def test_latent_mask_workflow_matches_omni_node_interfaces(workflow): + for node in workflow["nodes"]: + if not node["type"].startswith("VLLMOmni"): + continue + cls = getattr(omni_nodes, node["type"]) + schema = cls.INPUT_TYPES() + inputs = {**schema.get("required", {}), **schema.get("optional", {})} + for input_ in node["inputs"]: + kind = inputs[input_["name"]][0] + assert input_["type"] == ("COMBO" if isinstance(kind, list) else kind) + assert tuple(output["type"] for output in node["outputs"]) == cls.RETURN_TYPES + + +def test_latent_mask_workflow_defaults(workflow): + cases = _cases(workflow) + assert set(cases) == {"Object Removal", "Inpainting", "Continuation", "Extension"} + for generate in cases.values(): + assert generate["widgets_values"][0] == "http://127.0.0.1:8000/v1" + assert generate["widgets_values"][4:6] == [1344, 768] + + +@pytest.mark.parametrize("case", ["Object Removal", "Inpainting"]) +def test_spatial_cases_use_a_composited_mask(workflow, case): + edit = _source_of(workflow, _cases(workflow)[case], "latent_edit") + assert _source_of(workflow, edit, "source_video")["type"] == "LoadVideo" + assert _source_of(workflow, edit, "video_mask")["type"] == "MaskComposite" + + +@pytest.mark.parametrize("case", ["Continuation", "Extension"]) +def test_temporal_cases_build_a_per_frame_mask(workflow, case): + # The server pads a short [T, H, W] mask with its last slice, so the mask must + # come from the Temporal Mask node (one slice per output frame), not a + # hand-batched stack of a few slices. + generate = _cases(workflow)[case] + edit = _source_of(workflow, generate, "latent_edit") + temporal = _source_of(workflow, edit, "video_mask") + assert temporal["type"] == "VLLMOmniMiniMaxH3TemporalMask" + assert _source_of(workflow, temporal, "images")["type"] == "GetVideoComponents" + assert _source_of(workflow, edit, "source_video")["type"] == "LoadVideo" + _, duration, mode, _ = temporal["widgets_values"] + assert mode == case.lower() + assert duration == generate["widgets_values"][7]