From b78bca50ef5739d23f3f2595a075a39fd56e9d52 Mon Sep 17 00:00:00 2001 From: wangyu <410167048@qq.com> Date: Wed, 16 Sep 2026 14:41:59 +0800 Subject: [PATCH 01/15] Add lingbot perf test Signed-off-by: wangyu <410167048@qq.com> --- .buildkite/common/ci_source_file_dependencies.yml | 5 +++++ .buildkite/cuda/test-nightly.yml | 14 ++++++++++++++ .../perf/tests/test_lingbot_video_vllm_omni.json | 12 ++++++++++++ 3 files changed, 31 insertions(+) diff --git a/.buildkite/common/ci_source_file_dependencies.yml b/.buildkite/common/ci_source_file_dependencies.yml index 33def4de450..662d5cad635 100644 --- a/.buildkite/common/ci_source_file_dependencies.yml +++ b/.buildkite/common/ci_source_file_dependencies.yml @@ -435,6 +435,11 @@ source_file_dependencies: - tests/e2e/online_serving/test_lingbot_video_moe.py - tests/e2e/offline_inference/test_lingbot_world_v2.py + diffusion_lingbot_perf: + - *lingbot_video + - tests/dfx/perf/scripts/run_diffusion_benchmark.py + - tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json + # Invalid-param jobs: model business code only, plus the reliability scripts. reliability_invalid_param_h100: diff --git a/.buildkite/cuda/test-nightly.yml b/.buildkite/cuda/test-nightly.yml index 1fdb1ef4a3f..66be637a7b4 100644 --- a/.buildkite/cuda/test-nightly.yml +++ b/.buildkite/cuda/test-nightly.yml @@ -448,6 +448,20 @@ steps: buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" exit $$EXIT + - label: ":full_moon: Diffusion X2V · Perf Test · LingBot" + source_file_dependencies: diffusion_lingbot_perf + key: nightly-diffusion-x2v-performance-lingbot + timeout_in_minutes: 180 + commands: + - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results + - | + set +e + pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json -m "H100 and B200 and cards_1" + EXIT=$$? + buildkite-agent artifact upload "tests/dfx/perf/results/diffusion_result_*.json" + buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" + exit $$EXIT + - group: ":card_index_dividers: Diffusion Test" key: nightly-diffusion-test-group depends_on: upload-nightly-pipeline diff --git a/tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json b/tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json index 07403e78f9a..2249afa63a5 100644 --- a/tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json +++ b/tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json @@ -1,6 +1,18 @@ [ { "test_name": "test_lingbot_video_single_device", + "mark": [ + { + "hardware_marks": { + "res": { + "cuda": ["H100", "B200"] + }, + "num_cards": 1 + } + }, + "full_model", + "diffusion" + ], "description": "Single-device LingBot dense T2V baseline at 320x192, 9 frames, 2 steps.", "server_type": "vllm-omni", "server_params": { From e974cc2fe7305b4510995e2a6c652e12e368b322 Mon Sep 17 00:00:00 2001 From: wangyu <410167048@qq.com> Date: Thu, 17 Sep 2026 16:53:45 +0800 Subject: [PATCH 02/15] Refactor benchmark scripts to unify naming conventions and improve clarity - Updated references from `run_diffusion_benchmark.py` to `run_benchmark.py` in various CI configurations and test files to standardize the script used for performance benchmarks. - Adjusted test configurations to reflect changes in dataset naming and parameters, ensuring compatibility with the new benchmark script. - Enhanced documentation to clarify the distinction between diffusion and omni benchmarks, including updates to test examples and execution guides. This refactor aims to streamline the testing process and improve maintainability across the codebase. Signed-off-by: wangyu <410167048@qq.com> --- .../common/ci_source_file_dependencies.yml | 10 +- .buildkite/cuda/test-nightly.yml | 155 ++++------ .buildkite/cuda/test-weekly.yml | 30 +- .buildkite/npu/test-npu-nightly.yml | 7 +- .../test_examples/l4_performance_tests.inc.md | 52 ++-- docs/contributing/ci/test_execution_guide.md | 2 +- docs/contributing/ci/test_writing_guide.md | 3 +- recipes/LTX/LTX-2.md | 18 +- tests/dfx/conftest.py | 24 +- tests/dfx/perf/scripts/run_benchmark.py | 119 +++++++- .../perf/scripts/run_diffusion_benchmark.py | 16 +- .../perf/tests/test_cosmos3_vllm_omni.json | 185 ++++++------ .../test_hunyuanvideo15_i2v_vllm_omni.json | 100 ++++--- .../test_hunyuanvideo15_t2v_vllm_omni.json | 80 +++--- .../tests/test_lingbot_video_vllm_omni.json | 90 +++--- tests/dfx/perf/tests/test_ltx2_vllm_omni.json | 265 ++++++++++-------- .../perf/tests/test_minimax_h3_vllm_omni.json | 207 +++++++------- tests/dfx/perf/tests/test_runner_metadata.py | 40 ++- .../perf/tests/test_wan22_i2v_vllm_omni.json | 156 ++++++----- tools/nightly/run_nightly_jobs.sh | 15 +- 20 files changed, 903 insertions(+), 671 deletions(-) diff --git a/.buildkite/common/ci_source_file_dependencies.yml b/.buildkite/common/ci_source_file_dependencies.yml index 662d5cad635..6795e9678b7 100644 --- a/.buildkite/common/ci_source_file_dependencies.yml +++ b/.buildkite/common/ci_source_file_dependencies.yml @@ -296,7 +296,7 @@ source_file_dependencies: diffusion_wan22_perf: - *wan22 - - tests/dfx/perf/scripts/run_diffusion_benchmark.py + - tests/dfx/perf/scripts/run_benchmark.py - tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json diffusion_wan22_reliability: @@ -322,7 +322,7 @@ source_file_dependencies: diffusion_cosmos3_perf: - *cosmos3 - - tests/dfx/perf/scripts/run_diffusion_benchmark.py + - tests/dfx/perf/scripts/run_benchmark.py - tests/dfx/perf/tests/test_cosmos3_vllm_omni.json diffusion_qwen_image_function: @@ -395,7 +395,7 @@ source_file_dependencies: diffusion_minimax_h3_perf: - *minimax_h3 - - tests/dfx/perf/scripts/run_diffusion_benchmark.py + - tests/dfx/perf/scripts/run_benchmark.py - tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json diffusion_hunyuan_video_function: @@ -409,7 +409,7 @@ source_file_dependencies: diffusion_hunyuan_video_perf: - *hunyuan_video - - tests/dfx/perf/scripts/run_diffusion_benchmark.py + - tests/dfx/perf/scripts/run_benchmark.py - tests/dfx/perf/tests/test_hunyuanvideo15_t2v_vllm_omni.json - tests/dfx/perf/tests/test_hunyuanvideo15_i2v_vllm_omni.json @@ -437,7 +437,7 @@ source_file_dependencies: diffusion_lingbot_perf: - *lingbot_video - - tests/dfx/perf/scripts/run_diffusion_benchmark.py + - tests/dfx/perf/scripts/run_benchmark.py - tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json diff --git a/.buildkite/cuda/test-nightly.yml b/.buildkite/cuda/test-nightly.yml index 66be637a7b4..894d2047f09 100644 --- a/.buildkite/cuda/test-nightly.yml +++ b/.buildkite/cuda/test-nightly.yml @@ -53,80 +53,62 @@ steps: - label: ":full_moon: Omni · Accuracy Test" source_file_dependencies: omni_qwen3_omni_accuracy timeout_in_minutes: 180 + artifact_paths: + - tests/e2e/accuracy/qwen3_omni/results/qwen_omni_acc/*.json commands: - export SEED_TTS_WER_EVAL=1 - export SEED_TTS_EVAL_DEVICE=cuda:1 - - | - set +e - pytest -s -v tests/e2e/accuracy/qwen3_omni/test_qwen3_omni.py -m "full_model and H100 and B200 and cards_2" --run-level full_model - EXIT=$$? - buildkite-agent artifact upload "tests/e2e/accuracy/qwen3_omni/results/qwen_omni_acc/*.json" - exit $$EXIT + - pytest -s -v tests/e2e/accuracy/qwen3_omni/test_qwen3_omni.py -m "full_model and H100 and B200 and cards_2" --run-level full_model - label: ":full_moon: Omni · MiniCPM-o 4.5 · Accuracy Test" source_file_dependencies: omni_minicpmo_4_5_accuracy timeout_in_minutes: 180 + artifact_paths: + - tests/e2e/accuracy/minicpmo_4_5/results/*.json commands: - export SEED_TTS_WER_EVAL=1 - export SEED_TTS_EVAL_DEVICE=cuda:1 - - | - set +e - pytest -s -v tests/e2e/accuracy/minicpmo_4_5/test_minicpmo_4_5.py -m "full_model and H100 and B200 and cuda and (cards_1 or cards_2)" --run-level full_model - EXIT=$$? - buildkite-agent artifact upload "tests/e2e/accuracy/minicpmo_4_5/results/*.json" - exit $$EXIT + - pytest -s -v tests/e2e/accuracy/minicpmo_4_5/test_minicpmo_4_5.py -m "full_model and H100 and B200 and cuda and (cards_1 or cards_2)" --run-level full_model - label: ":full_moon: Omni · Perf Test · No Async Chunk" source_file_dependencies: omni_qwen3_omni_perf key: nightly-omni-performance-no-async-chunk timeout_in_minutes: 300 + artifact_paths: + - tests/dfx/perf/results/*.json commands: - export BENCHMARK_DIR=tests/dfx/perf/results - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_no_async_chunk.json -m "H100 and B200 and cards_2" - EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/*.json" - exit $$EXIT + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_no_async_chunk.json -m "H100 and B200 and cards_2" - label: ":full_moon: Omni · Perf Test · Async Chunk" source_file_dependencies: omni_qwen3_omni_perf key: nightly-omni-performance-async-chunk timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json commands: - export BENCHMARK_DIR=tests/dfx/perf/results - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_async_chunk.json -m "H100 and B200 and full_model and cards_2" - EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/*.json" - exit $$EXIT + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_async_chunk.json -m "H100 and B200 and full_model and cards_2" - label: ":full_moon: Omni · MiniCPM-o 4.5 · Perf Test" source_file_dependencies: omni_minicpmo_4_5_perf key: nightly-omni-performance-minicpmo-4-5 timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json commands: - export BENCHMARK_DIR=tests/dfx/perf/results - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minicpmo_4_5.json -m "H100 and B200 and cards_1" - EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/*.json" - exit $$EXIT + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minicpmo_4_5.json -m "H100 and B200 and cards_1" - label: ":full_moon: Omni · MiniCPM-o 4.5 · Duplex Seed-TTS Perf Test" source_file_dependencies: omni_minicpmo_4_5_duplex_perf key: nightly-omni-performance-minicpmo-4-5-duplex-seed-tts timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json commands: - export BENCHMARK_DIR=tests/dfx/perf/results - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minicpmo_4_5_duplex_seed_tts.json -m "H100 and B200 and cards_1" - EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/*.json" - exit $$EXIT + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minicpmo_4_5_duplex_seed_tts.json -m "H100 and B200 and cards_1" - label: ":full_moon: Omni · Multi-Replica Startup Test with 4x H100" source_file_dependencies: omni_qwen3_omni_function @@ -153,29 +135,23 @@ steps: source_file_dependencies: tts_qwen3_tts_perf key: nightly-tts-performance-single timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json commands: - export BENCHMARK_DIR=tests/dfx/perf/results - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_tts.json -m "H100 and B200 and cards_1" - EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/*.json" - exit $$EXIT + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_tts.json -m "H100 and B200 and cards_1" - label: ":full_moon: TTS · Perf Test · 2-GPU" source_file_dependencies: tts_qwen3_tts_perf key: nightly-tts-performance-2gpu timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json commands: - export BENCHMARK_DIR=tests/dfx/perf/results - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_tts.json -m "H100 and B200 and cards_2" - EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/*.json" - exit $$EXIT + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_tts.json -m "H100 and B200 and cards_2" - group: ":card_index_dividers: Tiny Model Tests [Multi-GPU]" key: nightly-tiny-model-test-group @@ -266,33 +242,27 @@ steps: source_file_dependencies: diffusion_qwen_image_perf key: nightly-diffusion-x2iat-performance-qwen-image-single timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/diffusion_result_*.json + - tests/dfx/perf/results/logs/*.log commands: - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results - export CACHE_DIT_VERSION=1.5.0 - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_1" - EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/diffusion_result_*.json" - buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" - exit $$EXIT + - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_1" - label: ":full_moon: Diffusion X2I(&A&T) · Perf Test · Qwen-Image · 4-GPU" source_file_dependencies: diffusion_qwen_image_perf key: nightly-diffusion-x2iat-performance-qwen-image-4gpu timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/diffusion_result_*.json + - tests/dfx/perf/results/logs/*.log commands: - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results - export CACHE_DIT_VERSION=1.5.0 # Do not pin DIFFUSION_ATTENTION_BACKEND: H100 auto-selects FLASH_ATTN; # B200 rejects explicit FLASH_ATTN without FA4 and uses CUDNN/TRTLLM. - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_4" - EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/diffusion_result_*.json" - buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" - exit $$EXIT + - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_4" - group: ":card_index_dividers: Diffusion X2V Model Test" key: nightly-diffusion-x2v-group @@ -388,15 +358,11 @@ steps: source_file_dependencies: diffusion_wan22_perf key: nightly-diffusion-x2v-performance-single timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json commands: - - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json -m "H100 and B200 and cards_1" - EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/diffusion_result_*.json" - buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" - exit $$EXIT + - export BENCHMARK_DIR=tests/dfx/perf/results + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json -m "H100 and B200 and cards_1" - label: ":full_moon: Diffusion X2V · Perf Test · 2-GPU" source_file_dependencies: @@ -404,16 +370,16 @@ steps: - diffusion_cosmos3_perf key: nightly-diffusion-x2v-performance-2gpu timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json commands: - - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results + - export BENCHMARK_DIR=tests/dfx/perf/results - | set +e - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json -m "H100 and B200 and cards_2" + pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json -m "H100 and B200 and cards_2" EXIT1=$$? - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_cosmos3_vllm_omni.json -m "H100 and B200 and cards_2" + pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_cosmos3_vllm_omni.json -m "H100 and B200 and cards_2" EXIT2=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/diffusion_result_*.json" - buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" if [ $$EXIT1 -ne 0 ]; then exit $$EXIT1; fi exit $$EXIT2 @@ -421,16 +387,16 @@ steps: source_file_dependencies: diffusion_hunyuan_video_perf key: nightly-diffusion-x2v-performance-hunyuanvideo15 timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json commands: - - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results + - export BENCHMARK_DIR=tests/dfx/perf/results - | set +e - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_hunyuanvideo15_t2v_vllm_omni.json -m "H100 and B200 and (cards_1 or cards_2)" + pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_hunyuanvideo15_t2v_vllm_omni.json -m "H100 and B200 and (cards_1 or cards_2)" EXIT1=$$? - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_hunyuanvideo15_i2v_vllm_omni.json -m "H100 and B200 and (cards_1 or cards_2)" + pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_hunyuanvideo15_i2v_vllm_omni.json -m "H100 and B200 and (cards_1 or cards_2)" EXIT2=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/diffusion_result_*.json" - buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" if [ $$EXIT1 -ne 0 ]; then exit $$EXIT1; fi exit $$EXIT2 @@ -438,29 +404,21 @@ steps: source_file_dependencies: diffusion_minimax_h3_perf key: nightly-diffusion-x2v-performance-minimax-h3 timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json commands: - - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json -m "H100 and B200 and cards_4" - EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/diffusion_result_*.json" - buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" - exit $$EXIT + - export BENCHMARK_DIR=tests/dfx/perf/results + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json -m "H100 and B200 and cards_4" - label: ":full_moon: Diffusion X2V · Perf Test · LingBot" source_file_dependencies: diffusion_lingbot_perf key: nightly-diffusion-x2v-performance-lingbot timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json commands: - - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json -m "H100 and B200 and cards_1" - EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/diffusion_result_*.json" - buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" - exit $$EXIT + - export BENCHMARK_DIR=tests/dfx/perf/results + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json -m "H100 and B200 and cards_1" - group: ":card_index_dividers: Diffusion Test" key: nightly-diffusion-test-group @@ -536,6 +494,9 @@ steps: - nightly-diffusion-x2v-performance-single - nightly-diffusion-x2v-performance-2gpu if: build.env("NIGHTLY") == "1" && build.env("EMAIL_DISTRIBUTION") == "1" + artifact_paths: + - tests/dfx/perf/results/*.xlsx + - tests/dfx/perf/results/*.html commands: - pip install openpyxl - export DEFAULT_INPUT_DIR=tests/dfx/perf/results @@ -551,7 +512,5 @@ steps: - python tools/nightly/generate_nightly_perf_excel.py - python tools/nightly/generate_nightly_perf_html.py - python tools/nightly/send_nightly_email.py --report-file "tests/dfx/perf/results/*.xlsx, tests/dfx/perf/results/*.html" - - buildkite-agent artifact upload "tests/dfx/perf/results/*.xlsx" - - buildkite-agent artifact upload "tests/dfx/perf/results/*.html" agents: queue: "cpu_queue_premerge" diff --git a/.buildkite/cuda/test-weekly.yml b/.buildkite/cuda/test-weekly.yml index 70ab5935673..5c193945e34 100644 --- a/.buildkite/cuda/test-weekly.yml +++ b/.buildkite/cuda/test-weekly.yml @@ -12,6 +12,8 @@ steps: depends_on: upload-weekly-pipeline if: build.env("WEEKLY") == "1" timeout_in_minutes: 60 + artifact_paths: + - coverage-core-model-cpu-*.xml.gz commands: - | set +e @@ -19,7 +21,6 @@ steps: pytest -sv tests/ -m 'core_model and cpu' --cov=vllm_omni --cov-report=term-missing:skip-covered --cov-report=xml:$${REPORT} EXIT=$$? gzip -9 -f "$${REPORT}" - buildkite-agent artifact upload "$${REPORT}.gz" exit $$EXIT mirror_hardwares: l4_1 @@ -79,41 +80,32 @@ steps: source_file_dependencies: omni_qwen3_omni_perf key: weekly-omni-performance-vllm-text timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json commands: - export BENCHMARK_DIR=tests/dfx/perf/results - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_vllm_text.json - EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/*.json" - exit $$EXIT + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_vllm_text.json mirror_hardwares: h100_2 - label: ":full_moon: Omni · Perf Test · Async Chunk · Random" key: weekly-omni-performance-async-chunk-random timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json commands: - export BENCHMARK_DIR=tests/dfx/perf/results - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_async_chunk.json -m "H100 and slow" - EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/*.json" - exit $$EXIT + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_async_chunk.json -m "H100 and slow" mirror_hardwares: h100_2 - label: ":full_moon: Omni · Perf Test · Multi-Replica" source_file_dependencies: omni_qwen3_omni_perf key: weekly-omni-performance-multi-replicas timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json commands: - export BENCHMARK_DIR=tests/dfx/perf/results - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_multi_replicas.json - EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/*.json" - exit $$EXIT + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_multi_replicas.json mirror_hardwares: h100_3 - group: ":card_index_dividers: E2E Tests" diff --git a/.buildkite/npu/test-npu-nightly.yml b/.buildkite/npu/test-npu-nightly.yml index f64fa711f64..4dfa2e77914 100644 --- a/.buildkite/npu/test-npu-nightly.yml +++ b/.buildkite/npu/test-npu-nightly.yml @@ -255,11 +255,10 @@ steps: HF_TOKEN: "${HF_TOKEN}" DIFFUSION_ATTENTION_BACKEND: TORCH_SDPA commands: - - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results + - export BENCHMARK_DIR=tests/dfx/perf/results - | set +e - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py -m npu --test-config-file tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json + pytest -s -v tests/dfx/perf/scripts/run_benchmark.py -m npu --test-config-file tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/diffusion_result_*.json" - buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" + buildkite-agent artifact upload "tests/dfx/perf/results/*.json" exit $$EXIT diff --git a/docs/contributing/ci/test_examples/l4_performance_tests.inc.md b/docs/contributing/ci/test_examples/l4_performance_tests.inc.md index feb4f38ede7..981e6089e3c 100644 --- a/docs/contributing/ci/test_examples/l4_performance_tests.inc.md +++ b/docs/contributing/ci/test_examples/l4_performance_tests.inc.md @@ -7,16 +7,17 @@ When you want to add L4-level ***performance test*** cases, add entries to JSON | Omni (nightly) | `run_benchmark.py` | `test_qwen3_omni_no_async_chunk.json`, `test_qwen3_omni_async_chunk.json` (`full_model` without `slow` in `mark`) | | Omni (weekly) | `run_benchmark.py` | `test_qwen3_omni_async_chunk.json` (CUDA only), `test_qwen3_omni_vllm_text.json`, `test_qwen3_omni_multi_replicas.json` (`slow` in `mark`; **Perf Test** in `test-weekly.yml`) | | TTS | `run_benchmark.py` | `test_tts.json`, `test_voxcpm2.json`, `test_higgs_audio_v3.json` | -| Diffusion | `run_diffusion_benchmark.py` | `test_qwen_image_vllm_omni.json`, `test_bagel_vllm_omni.json`, `test_wan22_i2v_vllm_omni.json`, `test_cosmos3_vllm_omni.json`, … | +| Diffusion (`/v1/chat/completions`) | `run_diffusion_benchmark.py` | `test_qwen_image_vllm_omni.json`, `test_bagel_vllm_omni.json`, … | +| Diffusion (`/v1/images/generations`, `/v1/images/edits`, `/v1/videos`) | `run_benchmark.py` | `test_wan22_i2v_vllm_omni.json`, `test_cosmos3_vllm_omni.json`, `test_lingbot_video_vllm_omni.json`, … | #### How runners pick cases Without **`--test-config-file`**, each runner scans all `*.json` under `tests/dfx/perf/tests/` but only keeps its own model type: -- **`run_benchmark.py`**: omni and TTS cases only (skips diffusion JSON). -- **`run_diffusion_benchmark.py`**: diffusion cases only (skips omni / TTS JSON). +- **`run_benchmark.py`**: cases whose `benchmark_params` use ``dataset_name`` (omni, TTS, and image/video OpenAI generation via `vllm bench serve --omni`). +- **`run_diffusion_benchmark.py`**: cases whose `benchmark_params` use ``dataset`` (diffusion client schema, including custom jsonl). -Diffusion cases are detected when the JSON has `server_type` (typically `"vllm-omni"`) or `"diffusion"` in the `mark` array. Omni / TTS JSON has neither. +Diffusion-script cases are detected by `is_diffusion_perf_config()`: presence of ``dataset`` (without ``dataset_name``) in `benchmark_params`. Marks / `server_type` / endpoint values are not used for this split. #### Running perf cases @@ -60,16 +61,16 @@ Pass **`--test-config-file`** to load one JSON file, or omit it for the bulk sca ##### Overview -| Field | Required | Description | -| ------------------ | -------------- | -------------------------------------------- | -| test_name | Yes | Unique identifier for the test case | -| mark | No | Pytest marks; see **`mark` field** below | -| server_params | Yes | Server-side configuration parameters | -| benchmark_params | Yes | Benchmark running parameters | -| server_type | Diffusion only | Routes case to run_diffusion_benchmark.py | -| benchmark_endpoint | Diffusion only | Benchmark API path | +| Field | Required | Description | +| ------------------ | --------- | ---------------------------------------------------------------------------------------- | +| test_name | Yes | Unique identifier for the test case | +| mark | No | Pytest marks; see **`mark` field** below | +| server_params | Yes | Server-side configuration parameters | +| benchmark_params | Yes | Benchmark running parameters | +| server_type | Diffusion | Only for diffusion-script JSON; omit on omni-bench generation cases | +| benchmark_endpoint | Optional | Legacy diffusion custom-jsonl alias; prefer `benchmark_params[].endpoint` | -Omit `mark` only for configs not meant to be filtered by `-m`. `server_type` is typically `"vllm-omni"`. `benchmark_endpoint` examples: `/v1/videos`, `/v1/images/generations`. +Omit `mark` only for configs not meant to be filtered by `-m`. Cases that call `/v1/images/edits`, `/v1/images/generations`, or `/v1/videos` use the same `benchmark_params` schema as Omni/TTS (`dataset_name`, `endpoint`, `extra_body`) and are executed by `run_benchmark.py` — do not set `server_type` or `task` on those cases. Remaining diffusion cases (usually `/v1/chat/completions`, or custom jsonl) stay on `run_diffusion_benchmark.py` and may keep `server_type`. #### `mark` field @@ -100,7 +101,7 @@ Recommended for L4 perf cases: } ``` -Multi-GPU diffusion (example: Cosmos3 with `cfg-parallel-size=2`): +Multi-GPU generation via omni bench (example: Cosmos3 with `cfg-parallel-size=2`): ```JSON { @@ -110,10 +111,17 @@ Multi-GPU diffusion (example: Cosmos3 with `cfg-parallel-size=2`): "full_model", "diffusion" ], - "server_type": "vllm-omni", - "benchmark_endpoint": "/v1/images/generations", "server_params": { "...": "..." }, - "benchmark_params": [ { "name": "1024x1024_steps4", "...": "..." } ] + "benchmark_params": [ + { + "name": "1024x1024_steps4", + "dataset_name": "random", + "endpoint": "/v1/images/generations", + "num_prompts": 3, + "max_concurrency": 1, + "extra_body": { "width": 1024, "height": 1024, "num_inference_steps": 4 } + } + ] } ``` @@ -132,8 +140,8 @@ Result files use the **runtime** hardware label from `get_runtime_resource_label Examples: -- Omni/TTS: `result_{test_name}_{optional_hw}_{dataset}_....json` under `BENCHMARK_DIR` -- Diffusion: one aggregate `diffusion_result_{config_stem}_{optional_hw}_{timestamp}.json` per source JSON file (array of all runs from that file) +- Omni/TTS and OpenAI generation endpoints (`/v1/images/*`, `/v1/videos`): `result_{test_name}_{optional_hw}_{dataset}_....json` under `BENCHMARK_DIR` +- Remaining diffusion (`run_diffusion_benchmark.py`): one aggregate `diffusion_result_{config_stem}_{optional_hw}_{timestamp}.json` per source JSON file #### Local commands @@ -145,6 +153,9 @@ pytest -s -v tests/dfx/perf/scripts/run_benchmark.py -m "full_model and omni and # Single file (same selectors as the CI Perf steps) pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py \ --test-config-file tests/dfx/perf/tests/test_bagel_vllm_omni.json +pytest -s -v tests/dfx/perf/scripts/run_benchmark.py \ + --test-config-file tests/dfx/perf/tests/test_cosmos3_vllm_omni.json \ + -m "H100 and B200 and cards_2" pytest -s -v tests/dfx/perf/scripts/run_benchmark.py \ --test-config-file tests/dfx/perf/tests/test_qwen3_omni_async_chunk.json \ -m "H100 and full_model and not slow" @@ -203,7 +214,8 @@ You can add any benchmark running parameters you need here. For all optional par 2. For boolean variables in the running parameters, modify them to forms such as ignore_eos: true/false and fill them into the JSON file. 3. Optionally add a `baseline` object (see **Baseline thresholds** below). If you omit `baseline` or leave it empty, the performance test still runs but does not assert metric thresholds from this field. 4. Set `"name"` on each `benchmark_params` entry for stable pytest ids and readable result keys. -5. The qps and concurrency modes are recommended to be mutually exclusive. For detailed explanations, see the table below: +5. Image/video generation cases (`/v1/images/generations`, `/v1/images/edits`, `/v1/videos`) use this same schema: set `endpoint` to the API path, put width/height/steps/frames in `extra_body`, use `dataset_name: random` for text-only inputs or `random-mm` when the request needs a synthetic image/video. Do not set `server_type` or `task`, and do not use `random-request-config` or kebab-case diffusion client fields. +6. The qps and concurrency modes are recommended to be mutually exclusive. For detailed explanations, see the table below: | Parameter | Type | Required | Example/Values | Description | | --------------- | ------------- | -------- | -------------------- | ------------------------------------ | diff --git a/docs/contributing/ci/test_execution_guide.md b/docs/contributing/ci/test_execution_guide.md index 7597f72e48f..774ebe41cc2 100644 --- a/docs/contributing/ci/test_execution_guide.md +++ b/docs/contributing/ci/test_execution_guide.md @@ -165,7 +165,7 @@ Failed jobs: 1/2 pytest -sv tests/dfx/perf/scripts/run_benchmark.py -m "full_model and tts and H100" pytest -sv tests/dfx/perf/scripts/run_diffusion_benchmark.py -m "full_model and diffusion and H100" pytest -sv tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_tts.json - pytest -sv tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_cosmos3_vllm_omni.json + pytest -sv tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_cosmos3_vllm_omni.json ``` Nightly **Perf Test** jobs in [``test-nightly.yml``](https://github.com/vllm-project/vllm-omni/blob/main/.buildkite/cuda/test-nightly.yml) use ``--test-config-file`` only (no ``-m``). Weekly **Perf Test** in [``test-weekly.yml``](https://github.com/vllm-project/vllm-omni/blob/main/.buildkite/cuda/test-weekly.yml) runs ``test_qwen3_omni_vllm_text.json`` and ``test_qwen3_omni_multi_replicas.json`` (JSON ``mark`` includes ``slow``). E2e L4 function tests use ``full_model`` + ``--run-level full_model``. Example: diff --git a/docs/contributing/ci/test_writing_guide.md b/docs/contributing/ci/test_writing_guide.md index 762863bb3e9..54a6629bf78 100644 --- a/docs/contributing/ci/test_writing_guide.md +++ b/docs/contributing/ci/test_writing_guide.md @@ -159,7 +159,8 @@ When `mark` is present, it must be an **array** with exactly one ``hardware_mark } ``` -- Local bulk load: `pytest -sv tests/dfx/perf/scripts/run_diffusion_benchmark.py -m "full_model and diffusion and H100"` +- Local bulk load: `pytest -sv tests/dfx/perf/scripts/run_benchmark.py -m "full_model and H100"` (omni/TTS and `/v1/images/*` + `/v1/videos` diffusion) +- Diffusion chat-completions remaining cases: `pytest -sv tests/dfx/perf/scripts/run_diffusion_benchmark.py -m "full_model and diffusion and H100"` - Nightly CI perf steps: `--test-config-file tests/dfx/perf/tests/test__vllm_omni.json` (file selects cases; no `-m`) - Result filenames use **runtime** GPU detection (`get_runtime_resource_label`); `H100` is omitted on the default CI pool diff --git a/recipes/LTX/LTX-2.md b/recipes/LTX/LTX-2.md index 5a6a049eeed..b84bebd6680 100644 --- a/recipes/LTX/LTX-2.md +++ b/recipes/LTX/LTX-2.md @@ -5,7 +5,7 @@ ## Pipelines | `--model-class-name` | Task | Required checkpoint repositories | -|---|---|---| +| --- | --- | --- | | `LTX2Pipeline` | LTX-2 one-stage T2V/I2V | `Lightricks/LTX-2` | | `LTX2TwoStagePipeline` | LTX-2 ordinary two-stage T2V/I2V | `Lightricks/LTX-2` | | `LTX2DistilledOneStagePipeline` | LTX-2 merged-distilled one-stage T2V/I2V | `rootonchair/LTX-2-19b-distilled` | @@ -53,7 +53,7 @@ pipe(req, image=image, prompt=prompt) The consolidation also removes these registry names without aliases: | Removed name | Replacement | -|---|---| +| --- | --- | | `LTX23Pipeline` | `LTX2Pipeline`; checkpoint metadata selects LTX-2.3 | | `LTX2ImageToVideoPipeline` | `LTX2Pipeline` with `image=` | | `LTX23ImageToVideoPipeline` | `LTX2Pipeline` with `image=`; checkpoint metadata selects LTX-2.3 | @@ -67,7 +67,7 @@ offline and serving entrypoints already use named fields and are unaffected. ## One-Stage Defaults | Parameter | LTX-2 | LTX-2.3 | -|---|---:|---:| +| --- | ---: | ---: | | Width × height | 768 × 512 | 768 × 512 | | Frames / frame rate | 121 / 24 | 121 / 24 | | Denoise steps | 40 | 30 | @@ -85,7 +85,7 @@ default to `121`. ## Two-Stage Defaults | Parameter | Ordinary | Full-distilled | -|---|---:|---:| +| --- | ---: | ---: | | Final width × height | 1536 × 1024 | 1536 × 1024 | | Stage 1 width × height | 768 × 512 | 768 × 512 | | Frames / frame rate | 121 / 24 | 121 / 24 | @@ -172,7 +172,7 @@ spatio-temporal guidance (STG), cross-modality guidance, and rescaling. Distilled stages and ordinary Stage 2 are fixed positive-only. | Parameter | Default | Effect | Alias | -|---|---:|---|---| +| --- | ---: | --- | --- | | `video_cfg_scale` | 3.0 | Video text CFG; `1.0` disables it | `video_cfg_guidance_scale` | | `audio_cfg_scale` | 7.0 | Audio text CFG; `1.0` disables it | `audio_cfg_guidance_scale` | | `video_stg_scale` | 1.0 | Video STG; `0.0` disables it | `video_stg_guidance_scale` | @@ -210,7 +210,7 @@ per denoise step: `cond`, `uncond`, `ptb` (STG), and `mod` (cross-modality). The useful balanced configurations are therefore: | `--cfg-parallel-size` | Passes per rank | Guidance-slot utilization | Notes | -|---:|---:|---:|---| +| ---: | ---: | ---: | --- | | `1` | 4 | 100% | Single-rank fused guidance batch | | `2` | 2 | 100% | Recommended two-rank configuration | | `4` | 1 | 100% | One guidance pass per rank | @@ -269,7 +269,7 @@ noted below. ### Complete `forward` Surface | Argument | Type/default | Meaning and constraints | -|---|---|---| +| --- | --- | --- | | `req` | `DiffusionRequestBatch`, required | Only positional argument; contains prompts and per-request sampling parameters. | | `image` | image or batch, `None` | Direct value wins over request images; no image selects T2V. I2V accepts one image per prompt, and a batch cannot mix T2V/I2V. | | `prompt` | string or list, `None` | Positive-text fallback; request prompts win. Mutually exclusive with `prompt_embeds`. | @@ -306,7 +306,7 @@ request prompt payload; LTX guidance fields live in sampling `extra_args`. ### Recipe-Specific Request Capabilities | Override | One-stage | Ordinary two-stage | Distilled two-stage | -|---|---|---|---| +| --- | --- | --- | --- | | Guidance | Supported | Stage 1 only; Stage 2 is positive-only | Fixed positive-only | | Negative prompt/embeddings | Supported | Supported by Stage 1 | Rejected | | `num_inference_steps` | Supported | Controls Stage 1; Stage 2 uses 3 | Fixed at 8 for Stage 1; Stage 2 uses 3 | @@ -357,7 +357,7 @@ bundled offline CLI do not currently expose `sigmas`. - The output audio sample rate comes from the loaded components and is not a request parameter. - For benchmarks, use `tests/dfx/perf/tests/test_ltx2_vllm_omni.json` with - `tests/dfx/perf/scripts/run_diffusion_benchmark.py`. + `tests/dfx/perf/scripts/run_benchmark.py`. - Ordinary and full-distilled two-stage T2V/I2V are supported; HQ execution remains out of scope. diff --git a/tests/dfx/conftest.py b/tests/dfx/conftest.py index 864fb7ab382..f67d2fe225d 100644 --- a/tests/dfx/conftest.py +++ b/tests/dfx/conftest.py @@ -1,3 +1,6 @@ +# SPDX-License-Identifier: Apache-2.0 +# SPDX-FileCopyrightText: Copyright contributors to the vLLM-Omni project + import json import os import re @@ -84,17 +87,20 @@ def resolve_pytest_marks(mark_field: Any) -> list[pytest.MarkDecorator]: raise ValueError(f"mark must be a list; got {type(mark_field).__name__}") -def _mark_names(mark_field: Any) -> set[str]: - if isinstance(mark_field, list): - return {str(item) for item in mark_field if isinstance(item, str)} - return set() +def is_diffusion_perf_config(cfg: dict[str, Any]) -> bool: + """True for perf JSON cases intended for ``run_diffusion_benchmark.py``. + Schema split (not marks / endpoints): -def is_diffusion_perf_config(cfg: dict[str, Any]) -> bool: - """True for perf JSON cases intended for ``run_diffusion_benchmark.py``.""" - if cfg.get("server_type") is not None: - return True - return "diffusion" in _mark_names(cfg.get("mark")) + - ``benchmark_params[].dataset`` → diffusion client (``run_diffusion_benchmark.py``) + - ``benchmark_params[].dataset_name`` → ``vllm bench serve --omni`` (``run_benchmark.py``) + """ + for params in cfg.get("benchmark_params") or []: + if not isinstance(params, dict): + continue + if "dataset" in params and "dataset_name" not in params: + return True + return False def _marks_by_test_name(configs: list[dict[str, Any]]) -> dict[str, list[pytest.MarkDecorator]]: diff --git a/tests/dfx/perf/scripts/run_benchmark.py b/tests/dfx/perf/scripts/run_benchmark.py index 94df2edc55e..97239988dfd 100644 --- a/tests/dfx/perf/scripts/run_benchmark.py +++ b/tests/dfx/perf/scripts/run_benchmark.py @@ -50,12 +50,19 @@ def _get_config_file_from_argv() -> str | None: _all_configs = load_benchmark_configs(config_dir=_PERF_TESTS_DIR) BENCHMARK_CONFIGS = [cfg for cfg in _all_configs if not is_diffusion_perf_config(cfg)] print( - f"No --test-config-file: loaded {len(BENCHMARK_CONFIGS)} omni/tts case(s) from " + f"No --test-config-file: loaded {len(BENCHMARK_CONFIGS)} omni/tts/generation case(s) from " f"{_PERF_TESTS_DIR}/*.json (skipped {len(_all_configs) - len(BENCHMARK_CONFIGS)} diffusion; " f"use -m to filter, e.g. -m tts)" ) else: - BENCHMARK_CONFIGS = load_benchmark_configs(CONFIG_FILE_PATH) + _loaded = load_benchmark_configs(CONFIG_FILE_PATH) + BENCHMARK_CONFIGS = [cfg for cfg in _loaded if not is_diffusion_perf_config(cfg)] + skipped = len(_loaded) - len(BENCHMARK_CONFIGS) + if skipped: + print( + f"--test-config-file: loaded {len(BENCHMARK_CONFIGS)} omni/tts/generation case(s); " + f"skipped {skipped} remaining diffusion case(s) (chat completions / custom jsonl)" + ) DEPLOY_CONFIGS_DIR = Path(__file__).parent.parent / "deploy" server_to_benchmark_mapping = create_test_parameter_mapping(BENCHMARK_CONFIGS) @@ -93,24 +100,107 @@ def close(self) -> None: stack.close() +# OmniServer defaults for flags not already present in JSON ``serve_args`` / +# ``extra_cli_args``. Add new (flag, value) pairs here rather than special-casing. +_OMNI_DEFAULT_SERVER_ARGS: tuple[tuple[str, str], ...] = ( + ("--stage-init-timeout", "600"), + ("--init-timeout", "900"), +) + + +def _cli_flag_names(cli_args: tuple[str, ...] | list[str]) -> set[str]: + """Return long-option names present in a flat CLI argv list.""" + names: set[str] = set() + for item in cli_args: + token = str(item) + if not token.startswith("--"): + continue + names.add(token.split("=", 1)[0]) + return names + + +def _merge_omni_default_server_args( + extra_cli_args: tuple[str, ...] | list[str], + *, + use_omni: bool, + defaults: tuple[tuple[str, str], ...] = _OMNI_DEFAULT_SERVER_ARGS, +) -> list[str]: + """Fill Omni defaults for flags not already set in JSON-derived CLI args. + + JSON ``serve_args`` / ``extra_cli_args`` win; only missing flags are appended. + """ + if not use_omni: + return [] + present = _cli_flag_names(extra_cli_args) + args: list[str] = [] + for flag, value in defaults: + if flag not in present: + args += [flag, value] + return args + + +def _omni_server_env() -> dict[str, str]: + """Writable video/image storage for ``/v1/videos`` and related generation APIs.""" + result_dir = Path(os.environ.get("BENCHMARK_DIR", "tests/dfx/perf/results")) + storage_path = Path(os.environ.get("VLLM_OMNI_STORAGE_PATH", str(result_dir / "storage"))) + storage_path.mkdir(parents=True, exist_ok=True) + return {"VLLM_OMNI_STORAGE_PATH": str(storage_path)} + + +def _resolve_offline_model(model: str) -> str: + """Resolve HF ids / MiniMax env overrides the same way as the diffusion runner.""" + import huggingface_hub + + from vllm_omni.transformers_utils.repo_utils import hf_api + + if not model or os.path.isdir(model): + return model + + model_env_overrides = { + "MiniMaxAI/MiniMax-H3": "VLLM_TEST_MINIMAX_H3_MODEL", + "MiniMaxAI/MiniMax-H3/FL2VA": "VLLM_TEST_MINIMAX_H3_FL2VA_MODEL", + "MiniMaxAI/MiniMax-H3/Ref2VA": "VLLM_TEST_MINIMAX_H3_REF2VA_MODEL", + } + env_name = model_env_overrides.get(model) + if env_name: + env_model = os.environ.get(env_name) + if env_model: + return env_model + + parts = model.split("/") + if len(parts) >= 3: + repo_id = "/".join(parts[:2]) + subfolder = "/".join(parts[2:]) + snapshot_root = hf_api().snapshot_download( + repo_id, + allow_patterns=[f"{subfolder}/**"], + local_files_only=huggingface_hub.constants.HF_HUB_OFFLINE, + ) + return str(Path(snapshot_root) / subfolder) + + if not huggingface_hub.constants.HF_HUB_OFFLINE: + return model + return hf_api().snapshot_download(model, local_files_only=True) + + @contextmanager def _start_omni_server(server_param): test_name, model, stage_config_path, stage_overrides, extra_cli_args, use_omni = server_param + extra = tuple(extra_cli_args or ()) + model = _resolve_offline_model(model) print(f"Starting OmniServer with test: {test_name}, model: {model}") - server_args: list[str] = [] - if use_omni: - server_args += ["--stage-init-timeout", "600", "--init-timeout", "900"] + server_args: list[str] = _merge_omni_default_server_args(extra, use_omni=use_omni) # --deploy-config and --stage-overrides compose at the CLI (see vllm_omni/entrypoints/utils.py): # deploy-config sets the base; stage-overrides are applied on top. Both can be set. if stage_config_path: server_args = ["--deploy-config", stage_config_path] + server_args if stage_overrides: server_args = ["--stage-overrides", stage_overrides] + server_args - if extra_cli_args: - server_args = list(extra_cli_args) + server_args - with OmniServer(model, server_args, use_omni=use_omni) as server: + if extra: + server_args = list(extra) + server_args + with OmniServer(model, server_args, use_omni=use_omni, env_dict=_omni_server_env()) as server: server.test_name = test_name print("OmniServer started successfully") yield server @@ -259,11 +349,24 @@ def to_list(value, default=None): "eval_phase", "trust_remote_code", "expected_duplex_audio_turns_per_session", + "name", + "enable_negative_prompt", + "random_request_config", + "num_input_images", + "warmup_requests", + "warmup_concurrency", + "warmup_num_inference_steps", } + param_keys = {str(key).replace("-", "_") for key in params} + if "model" not in param_keys: + args.extend(["--model", str(model)]) + for key, value in params.items(): if key in exclude_keys or value is None: continue + if key in {"extra_body", "extra-body"} and value == {}: + continue arg_name = f"--{key.replace('_', '-')}" diff --git a/tests/dfx/perf/scripts/run_diffusion_benchmark.py b/tests/dfx/perf/scripts/run_diffusion_benchmark.py index ffcfc45b721..0523b9b3a1a 100644 --- a/tests/dfx/perf/scripts/run_diffusion_benchmark.py +++ b/tests/dfx/perf/scripts/run_diffusion_benchmark.py @@ -204,11 +204,11 @@ def load_diffusion_benchmark_configs( ) -> list[dict[str, Any]]: """Load one diffusion benchmark JSON, or merge all ``*.json`` under *config_dir*.""" if config_path is not None: - configs = load_configs(config_path) + loaded = load_configs(config_path) source = str(Path(config_path).resolve()) - for cfg in configs: + for cfg in loaded: cfg.setdefault(_DIFFUSION_SOURCE_CONFIG_KEY, source) - return configs + return loaded if config_dir is None: raise ValueError("load_diffusion_benchmark_configs requires config_path or config_dir") configs: list[dict[str, Any]] = [] @@ -231,7 +231,15 @@ def load_diffusion_benchmark_configs( f"use -m to filter, e.g. -m diffusion)" ) else: - BENCHMARK_CONFIGS = load_diffusion_benchmark_configs(CONFIG_FILE_PATH) + _loaded = load_diffusion_benchmark_configs(CONFIG_FILE_PATH) + BENCHMARK_CONFIGS = [cfg for cfg in _loaded if is_diffusion_perf_config(cfg)] + skipped = len(_loaded) - len(BENCHMARK_CONFIGS) + if skipped: + print( + f"--test-config-file: loaded {len(BENCHMARK_CONFIGS)} diffusion case(s); " + f"skipped {skipped} omni-bench generation case(s) " + f"(/v1/images/edits, /v1/images/generations, /v1/videos → run_benchmark.py)" + ) _AGGREGATED_RESULT_FILES_BY_SOURCE: dict[str, Path] = {} diff --git a/tests/dfx/perf/tests/test_cosmos3_vllm_omni.json b/tests/dfx/perf/tests/test_cosmos3_vllm_omni.json index 5e2b9d3dc7f..06bf70d50c7 100644 --- a/tests/dfx/perf/tests/test_cosmos3_vllm_omni.json +++ b/tests/dfx/perf/tests/test_cosmos3_vllm_omni.json @@ -5,7 +5,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 2 } @@ -14,8 +17,6 @@ "diffusion" ], "description": "Cosmos3-Nano text-to-image official demo workload on 2 GPUs", - "server_type": "vllm-omni", - "benchmark_endpoint": "/v1/images/generations", "server_params": { "model": "nvidia/Cosmos3-Nano", "serve_args": { @@ -34,24 +35,27 @@ "benchmark_params": [ { "name": "1024x1024_steps4", - "dataset": "random", - "task": "t2i", - "width": 1024, - "height": 1024, - "num-inference-steps": 4, - "num-prompts": 3, - "max-concurrency": 1, - "seed": 42, - "extra-body": { + "dataset_name": "random", + "endpoint": "/v1/images/generations", + "num_prompts": 3, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 1024, + "height": 1024, + "num_inference_steps": 4, + "seed": 42, "guidance_scale": 7.0, "flow_shift": 3.0, "guardrails": false }, "baseline": { "H100": { - "throughput_qps": 1.1958, - "latency_mean": 0.843, - "peak_memory_mb_mean": 80000.0 + "request_throughput": 1.1958, + "mean_e2el_ms": 843.0, + "mean_peak_memory_mb": 80000.0 } } } @@ -63,7 +67,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 2 } @@ -72,8 +79,6 @@ "diffusion" ], "description": "Cosmos3-Nano text-to-video official demo workload on 2 GPUs", - "server_type": "vllm-omni", - "benchmark_endpoint": "/v1/videos", "server_params": { "model": "nvidia/Cosmos3-Nano", "serve_args": { @@ -92,23 +97,20 @@ "benchmark_params": [ { "name": "1280x720_frames189_steps4", - "dataset": "random", - "task": "t2v", - "num-prompts": 3, - "max-concurrency": 1, - "seed": 42, - "random-request-config": [ - { - "width": 1280, - "height": 720, - "prompt": "A robot arm is cleaning a plate in the kitchen.", - "num_inference_steps": 4, - "num_frames": 189, - "fps": 24, - "weight": 1 - } - ], - "extra-body": { + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 1280, + "height": 720, + "num_inference_steps": 4, + "num_frames": 189, + "fps": 24, + "seed": 42, "guidance_scale": 6.0, "flow_shift": 10.0, "max_sequence_length": 4096, @@ -118,9 +120,9 @@ }, "baseline": { "H100": { - "throughput_qps": 0.0307, - "latency_mean": 32.7665, - "peak_memory_mb_mean": 28762.4762 + "request_throughput": 0.0307, + "mean_e2el_ms": 32766.5, + "mean_peak_memory_mb": 28762.4762 } } } @@ -132,7 +134,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 2 } @@ -141,8 +146,6 @@ "diffusion" ], "description": "Cosmos3-Nano image-to-video official demo workload on 2 GPUs", - "server_type": "vllm-omni", - "benchmark_endpoint": "/v1/videos", "server_params": { "model": "nvidia/Cosmos3-Nano", "serve_args": { @@ -161,24 +164,29 @@ "benchmark_params": [ { "name": "1280x720_frames189_steps4", - "dataset": "random", - "task": "i2v", - "num-prompts": 3, - "max-concurrency": 1, - "num-input-images": 1, - "seed": 42, - "random-request-config": [ - { - "width": 1280, - "height": 720, - "prompt": "The scene comes to life with smooth, natural motion.", - "num_inference_steps": 4, - "num_frames": 189, - "fps": 24, - "weight": 1 - } - ], - "extra-body": { + "dataset_name": "random-mm", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "random_range_ratio": 0.0, + "percentile_metrics": "e2el", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(720, 1280, 1)": 1.0 + }, + "extra_body": { + "width": 1280, + "height": 720, + "num_inference_steps": 4, + "num_frames": 189, + "fps": 24, + "seed": 42, "guidance_scale": 6.0, "flow_shift": 10.0, "max_sequence_length": 4096, @@ -186,9 +194,9 @@ }, "baseline": { "H100": { - "throughput_qps": 0.0305, - "latency_mean": 33.1009, - "peak_memory_mb_mean": 29441.1429 + "request_throughput": 0.0305, + "mean_e2el_ms": 33100.9, + "mean_peak_memory_mb": 29441.1429 } } } @@ -200,7 +208,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 2 } @@ -209,8 +220,6 @@ "diffusion" ], "description": "Cosmos3-Nano video-to-video official demo workload on 2 GPUs", - "server_type": "vllm-omni", - "benchmark_endpoint": "/v1/videos", "server_params": { "model": "nvidia/Cosmos3-Nano", "serve_args": { @@ -229,23 +238,29 @@ "benchmark_params": [ { "name": "1280x720_frames189_steps4", - "dataset": "random", - "task": "v2v", - "num-prompts": 3, - "max-concurrency": 1, - "seed": 42, - "random-request-config": [ - { - "width": 1280, - "height": 720, - "prompt": "Continue the same scene with smooth natural motion.", - "num_inference_steps": 4, - "num_frames": 189, - "fps": 24, - "weight": 1 - } - ], - "extra-body": { + "dataset_name": "random-mm", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "random_range_ratio": 0.0, + "percentile_metrics": "e2el", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "video": 1 + }, + "random_mm_bucket_config": { + "(720, 1280, 189)": 1.0 + }, + "extra_body": { + "width": 1280, + "height": 720, + "num_inference_steps": 4, + "num_frames": 189, + "fps": 24, + "seed": 42, "guidance_scale": 6.0, "flow_shift": 10.0, "max_sequence_length": 4096, @@ -258,9 +273,9 @@ }, "baseline": { "H100": { - "throughput_qps": 0.0299, - "latency_mean": 33.925, - "peak_memory_mb_mean": 29443.1429 + "request_throughput": 0.0299, + "mean_e2el_ms": 33925.0, + "mean_peak_memory_mb": 29443.1429 } } } diff --git a/tests/dfx/perf/tests/test_hunyuanvideo15_i2v_vllm_omni.json b/tests/dfx/perf/tests/test_hunyuanvideo15_i2v_vllm_omni.json index 25d91c44fda..925ccb23941 100644 --- a/tests/dfx/perf/tests/test_hunyuanvideo15_i2v_vllm_omni.json +++ b/tests/dfx/perf/tests/test_hunyuanvideo15_i2v_vllm_omni.json @@ -5,7 +5,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 1 } @@ -14,7 +17,6 @@ "diffusion" ], "description": "Single-device baseline", - "server_type": "vllm-omni", "server_params": { "model": "hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-480p_i2v", "serve_args": { @@ -25,26 +27,32 @@ "benchmark_params": [ { "name": "832x480_frames33_steps4", - "dataset": "random", - "task": "i2v", - "num-prompts": 10, - "max-concurrency": 1, - "num-input-images": 1, - "seed": 42, - "enable-negative-prompt": true, - "random-request-config": [ - { - "width": 832, - "height": 480, - "num_inference_steps": 4, - "num_frames": 33, - "fps": 24, - "weight": 1 - } - ], - "extra-body": { + "dataset_name": "random-mm", + "endpoint": "/v1/videos", + "num_prompts": 10, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "random_range_ratio": 0.0, + "percentile_metrics": "e2el", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(480, 832, 1)": 1.0 + }, + "extra_body": { + "width": 832, + "height": 480, + "num_inference_steps": 4, + "num_frames": 33, + "fps": 24, + "seed": 42, "guidance_scale": 6.0, - "flow_shift": 5.0 + "flow_shift": 5.0, + "negative_prompt": "Negative prompt for benchmarking diffusion models" } } ] @@ -55,7 +63,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 2 } @@ -64,7 +75,6 @@ "diffusion" ], "description": "CacheDiT + TP=2 + VAE patch parallel=2 + VAE tiling", - "server_type": "vllm-omni", "server_params": { "model": "hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-480p_i2v", "serve_args": { @@ -79,26 +89,32 @@ "benchmark_params": [ { "name": "832x480_frames33_steps4", - "dataset": "random", - "task": "i2v", - "num-prompts": 10, - "max-concurrency": 1, - "num-input-images": 1, - "seed": 42, - "enable-negative-prompt": true, - "random-request-config": [ - { - "width": 832, - "height": 480, - "num_inference_steps": 4, - "num_frames": 33, - "fps": 24, - "weight": 1 - } - ], - "extra-body": { + "dataset_name": "random-mm", + "endpoint": "/v1/videos", + "num_prompts": 10, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "random_range_ratio": 0.0, + "percentile_metrics": "e2el", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(480, 832, 1)": 1.0 + }, + "extra_body": { + "width": 832, + "height": 480, + "num_inference_steps": 4, + "num_frames": 33, + "fps": 24, + "seed": 42, "guidance_scale": 6.0, - "flow_shift": 5.0 + "flow_shift": 5.0, + "negative_prompt": "Negative prompt for benchmarking diffusion models" } } ] diff --git a/tests/dfx/perf/tests/test_hunyuanvideo15_t2v_vllm_omni.json b/tests/dfx/perf/tests/test_hunyuanvideo15_t2v_vllm_omni.json index ffaaba43614..65c2f85739a 100644 --- a/tests/dfx/perf/tests/test_hunyuanvideo15_t2v_vllm_omni.json +++ b/tests/dfx/perf/tests/test_hunyuanvideo15_t2v_vllm_omni.json @@ -5,7 +5,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 1 } @@ -14,7 +17,6 @@ "diffusion" ], "description": "Single-device baseline", - "server_type": "vllm-omni", "server_params": { "model": "hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-480p_t2v", "serve_args": { @@ -25,25 +27,23 @@ "benchmark_params": [ { "name": "832x480_frames33_steps4", - "dataset": "random", - "task": "t2v", - "num-prompts": 10, - "max-concurrency": 1, - "seed": 42, - "enable-negative-prompt": true, - "random-request-config": [ - { - "width": 832, - "height": 480, - "num_inference_steps": 4, - "num_frames": 33, - "fps": 24, - "weight": 1 - } - ], - "extra-body": { + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 10, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 832, + "height": 480, + "num_inference_steps": 4, + "num_frames": 33, + "fps": 24, + "seed": 42, "guidance_scale": 6.0, - "flow_shift": 5.0 + "flow_shift": 5.0, + "negative_prompt": "Negative prompt for benchmarking diffusion models" } } ] @@ -54,7 +54,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 2 } @@ -63,7 +66,6 @@ "diffusion" ], "description": "CacheDiT + TP=2 + VAE patch parallel=2 + VAE tiling", - "server_type": "vllm-omni", "server_params": { "model": "hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-480p_t2v", "serve_args": { @@ -78,25 +80,23 @@ "benchmark_params": [ { "name": "832x480_frames33_steps4", - "dataset": "random", - "task": "t2v", - "num-prompts": 10, - "max-concurrency": 1, - "seed": 42, - "enable-negative-prompt": true, - "random-request-config": [ - { - "width": 832, - "height": 480, - "num_inference_steps": 4, - "num_frames": 33, - "fps": 24, - "weight": 1 - } - ], - "extra-body": { + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 10, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 832, + "height": 480, + "num_inference_steps": 4, + "num_frames": 33, + "fps": 24, + "seed": 42, "guidance_scale": 6.0, - "flow_shift": 5.0 + "flow_shift": 5.0, + "negative_prompt": "Negative prompt for benchmarking diffusion models" } } ] diff --git a/tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json b/tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json index 2249afa63a5..88667e12f9e 100644 --- a/tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json +++ b/tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json @@ -1,45 +1,49 @@ [ - { - "test_name": "test_lingbot_video_single_device", - "mark": [ - { - "hardware_marks": { - "res": { - "cuda": ["H100", "B200"] - }, - "num_cards": 1 - } - }, - "full_model", - "diffusion" - ], - "description": "Single-device LingBot dense T2V baseline at 320x192, 9 frames, 2 steps.", - "server_type": "vllm-omni", - "server_params": { - "model": "robbyant/lingbot-video-dense-1.3b", - "serve_args": { - "model-class-name": "LingBotVideoPipeline", - "enable-diffusion-pipeline-profiler": true - } - }, - "benchmark_params": [ - { - "name": "320x192_frames9_steps2", - "dataset": "random", - "task": "t2v", - "backend": "v1/videos", - "width": 320, - "height": 192, - "num-frames": 9, - "fps": 24, - "num-inference-steps": 2, - "num-prompts": 3, - "max-concurrency": 1, - "warmup-requests": 1, - "warmup-concurrency": 1, - "warmup-num-inference-steps": 2, - "enable-negative-prompt": true - } - ] - } + { + "test_name": "test_lingbot_video_single_device", + "mark": [ + { + "hardware_marks": { + "res": { + "cuda": [ + "H100", + "B200" + ] + }, + "num_cards": 1 + } + }, + "full_model", + "diffusion" + ], + "description": "Single-device LingBot dense T2V baseline at 320x192, 9 frames, 2 steps.", + "server_params": { + "model": "robbyant/lingbot-video-dense-1.3b", + "serve_args": { + "model-class-name": "LingBotVideoPipeline", + "enable-diffusion-pipeline-profiler": true + } + }, + "benchmark_params": [ + { + "name": "320x192_frames9_steps2", + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "num_warmups": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 320, + "height": 192, + "num_frames": 9, + "fps": 24, + "num_inference_steps": 2, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + } + } + ] + } ] diff --git a/tests/dfx/perf/tests/test_ltx2_vllm_omni.json b/tests/dfx/perf/tests/test_ltx2_vllm_omni.json index 9b0634a202d..c5f986981fb 100644 --- a/tests/dfx/perf/tests/test_ltx2_vllm_omni.json +++ b/tests/dfx/perf/tests/test_ltx2_vllm_omni.json @@ -14,7 +14,6 @@ "diffusion" ], "description": "Single-device baseline with enforce-eager (no torch.compile)", - "server_type": "vllm-omni", "server_params": { "model": "Lightricks/LTX-2", "serve_args": { @@ -25,31 +24,39 @@ "benchmark_params": [ { "name": "256x256_145f_steps6", - "dataset": "random", - "task": "t2v", - "backend": "v1/videos", - "width": 256, - "height": 256, - "num-frames": 145, - "fps": 24, - "num-inference-steps": 6, - "num-prompts": 3, - "max-concurrency": 1, - "enable-negative-prompt": true + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 256, + "height": 256, + "num_frames": 145, + "fps": 24, + "num_inference_steps": 6, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + } }, { "name": "480x768_41f_steps20", - "dataset": "random", - "task": "t2v", - "backend": "v1/videos", - "width": 768, - "height": 480, - "num-frames": 41, - "fps": 24, - "num-inference-steps": 20, - "num-prompts": 3, - "max-concurrency": 1, - "enable-negative-prompt": true + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 768, + "height": 480, + "num_frames": 41, + "fps": 24, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + } } ] }, @@ -68,7 +75,6 @@ "diffusion" ], "description": "Single-device with torch.compile (default, no enforce-eager)", - "server_type": "vllm-omni", "server_params": { "model": "Lightricks/LTX-2", "serve_args": { @@ -78,31 +84,39 @@ "benchmark_params": [ { "name": "256x256_145f_steps6", - "dataset": "random", - "task": "t2v", - "backend": "v1/videos", - "width": 256, - "height": 256, - "num-frames": 145, - "fps": 24, - "num-inference-steps": 6, - "num-prompts": 3, - "max-concurrency": 1, - "enable-negative-prompt": true + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 256, + "height": 256, + "num_frames": 145, + "fps": 24, + "num_inference_steps": 6, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + } }, { "name": "480x768_41f_steps20", - "dataset": "random", - "task": "t2v", - "backend": "v1/videos", - "width": 768, - "height": 480, - "num-frames": 41, - "fps": 24, - "num-inference-steps": 20, - "num-prompts": 3, - "max-concurrency": 1, - "enable-negative-prompt": true + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 768, + "height": 480, + "num_frames": 41, + "fps": 24, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + } } ] }, @@ -121,7 +135,6 @@ "diffusion" ], "description": "CFG-parallel=2 with enforce-eager", - "server_type": "vllm-omni", "server_params": { "model": "Lightricks/LTX-2", "serve_args": { @@ -133,31 +146,39 @@ "benchmark_params": [ { "name": "256x256_145f_steps6", - "dataset": "random", - "task": "t2v", - "backend": "v1/videos", - "width": 256, - "height": 256, - "num-frames": 145, - "fps": 24, - "num-inference-steps": 6, - "num-prompts": 3, - "max-concurrency": 1, - "enable-negative-prompt": true + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 256, + "height": 256, + "num_frames": 145, + "fps": 24, + "num_inference_steps": 6, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + } }, { "name": "480x768_41f_steps20", - "dataset": "random", - "task": "t2v", - "backend": "v1/videos", - "width": 768, - "height": 480, - "num-frames": 41, - "fps": 24, - "num-inference-steps": 20, - "num-prompts": 3, - "max-concurrency": 1, - "enable-negative-prompt": true + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 768, + "height": 480, + "num_frames": 41, + "fps": 24, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + } } ] }, @@ -176,7 +197,6 @@ "diffusion" ], "description": "CFG-parallel=2 with torch.compile", - "server_type": "vllm-omni", "server_params": { "model": "Lightricks/LTX-2", "serve_args": { @@ -187,31 +207,39 @@ "benchmark_params": [ { "name": "256x256_145f_steps6", - "dataset": "random", - "task": "t2v", - "backend": "v1/videos", - "width": 256, - "height": 256, - "num-frames": 145, - "fps": 24, - "num-inference-steps": 6, - "num-prompts": 3, - "max-concurrency": 1, - "enable-negative-prompt": true + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 256, + "height": 256, + "num_frames": 145, + "fps": 24, + "num_inference_steps": 6, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + } }, { "name": "480x768_41f_steps20", - "dataset": "random", - "task": "t2v", - "backend": "v1/videos", - "width": 768, - "height": 480, - "num-frames": 41, - "fps": 24, - "num-inference-steps": 20, - "num-prompts": 3, - "max-concurrency": 1, - "enable-negative-prompt": true + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 768, + "height": 480, + "num_frames": 41, + "fps": 24, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + } } ] }, @@ -230,7 +258,6 @@ "diffusion" ], "description": "CacheDiT with enforce-eager", - "server_type": "vllm-omni", "server_params": { "model": "Lightricks/LTX-2", "serve_args": { @@ -242,31 +269,39 @@ "benchmark_params": [ { "name": "256x256_145f_steps6", - "dataset": "random", - "task": "t2v", - "backend": "v1/videos", - "width": 256, - "height": 256, - "num-frames": 145, - "fps": 24, - "num-inference-steps": 6, - "num-prompts": 3, - "max-concurrency": 1, - "enable-negative-prompt": true + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 256, + "height": 256, + "num_frames": 145, + "fps": 24, + "num_inference_steps": 6, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + } }, { "name": "480x768_41f_steps20", - "dataset": "random", - "task": "t2v", - "backend": "v1/videos", - "width": 768, - "height": 480, - "num-frames": 41, - "fps": 24, - "num-inference-steps": 20, - "num-prompts": 3, - "max-concurrency": 1, - "enable-negative-prompt": true + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 768, + "height": 480, + "num_frames": 41, + "fps": 24, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + } } ] } diff --git a/tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json b/tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json index 7dfbe6f3de9..dfd7d815f95 100644 --- a/tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json +++ b/tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json @@ -5,7 +5,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 4 } @@ -14,8 +17,6 @@ "diffusion" ], "description": "MiniMax-H3 T2VA text-to-video+audio on 4 GPUs (USP4 + HSDP4, encoder-TP4, VAE-PP4 tile)", - "server_type": "vllm-omni", - "benchmark_endpoint": "/v1/videos", "server_params": { "model": "MiniMaxAI/MiniMax-H3", "serve_args": { @@ -38,33 +39,29 @@ "benchmark_params": [ { "name": "1344x768_frames209_steps8", - "dataset": "random", - "task": "t2v", - "num-prompts": 3, - "max-concurrency": 1, - "warmup-requests": 1, - "warmup-concurrency": 1, - "seed": 42, - "random-request-config": [ - { - "width": 1344, - "height": 768, - "prompt": "Cinematic drone shot pushing through misty mountain ridges at sunrise, golden light catching the clouds, slow and smooth.", - "num_inference_steps": 8, - "num_frames": 209, - "fps": 24, - "weight": 1 - } - ], - "extra-body": { + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "num_warmups": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 1344, + "height": 768, + "num_inference_steps": 8, + "num_frames": 209, + "fps": 24, + "seed": 42, "flow_shift": 12.0, "extra_params": "{\"task\":\"t2va\",\"duration\":8.7,\"aspect_ratio\":\"16:9\",\"audio_flow_shift\":3.0}" }, "baseline": { "H100": { - "throughput_qps": 0.0211, - "latency_mean": 38.3252, - "peak_memory_mb_mean": 54219.6 + "request_throughput": 0.0211, + "mean_e2el_ms": 38325.2, + "mean_peak_memory_mb": 54219.6 } } } @@ -76,7 +73,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 4 } @@ -85,8 +85,6 @@ "diffusion" ], "description": "MiniMax-H3 FL2VA first-frame image-to-video+audio on 4 GPUs (USP4 + HSDP4, encoder-TP4, VAE-PP4 tile)", - "server_type": "vllm-omni", - "benchmark_endpoint": "/v1/videos", "server_params": { "model": "MiniMaxAI/MiniMax-H3", "serve_args": { @@ -109,34 +107,38 @@ "benchmark_params": [ { "name": "1344x768_frames209_steps8", - "dataset": "random", - "task": "ti2v", - "num-prompts": 3, - "max-concurrency": 1, - "warmup-requests": 1, - "warmup-concurrency": 1, - "seed": 42, - "random-request-config": [ - { - "width": 1344, - "height": 768, - "prompt": "Continue the scene: the camera slowly pushes in on the subject as the background city lights drift and blur, with the actor singing softly and emotionally.", - "num_inference_steps": 8, - "num_frames": 209, - "fps": 24, - "weight": 1 - } - ], - "extra-body": { + "dataset_name": "random-mm", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "num_warmups": 1, + "random_input_len": 64, + "random_output_len": 1, + "random_range_ratio": 0.0, + "percentile_metrics": "e2el", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(768, 1344, 1)": 1.0 + }, + "extra_body": { + "width": 1344, + "height": 768, + "num_inference_steps": 8, + "num_frames": 209, + "fps": 24, + "seed": 42, "flow_shift": 12.0, "extra_params": "{\"task\":\"fl2va\",\"duration\":8.7,\"audio_flow_shift\":3.0}" }, - "num-input-images": 1, "baseline": { "H100": { - "throughput_qps": 0.0212, - "latency_mean": 38.1385, - "peak_memory_mb_mean": 54622.8 + "request_throughput": 0.0212, + "mean_e2el_ms": 38138.5, + "mean_peak_memory_mb": 54622.8 } } } @@ -148,7 +150,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 4 } @@ -157,8 +162,6 @@ "diffusion" ], "description": "MiniMax-H3 Ref2VA video-reference-to-video+audio on 4 GPUs (USP4 + HSDP4, encoder-TP4, VAE-PP4 tile)", - "server_type": "vllm-omni", - "benchmark_endpoint": "/v1/videos", "server_params": { "model": "MiniMaxAI/MiniMax-H3", "serve_args": { @@ -181,33 +184,38 @@ "benchmark_params": [ { "name": "1344x768_frames209_steps8", - "dataset": "random", - "task": "v2v", - "num-prompts": 3, - "max-concurrency": 1, - "warmup-requests": 1, - "warmup-concurrency": 1, - "seed": 42, - "random-request-config": [ - { - "width": 1344, - "height": 768, - "prompt": "Restyle the reference footage into a cinematic night scene with neon reflections, keeping the same camera motion and subject.", - "num_inference_steps": 8, - "num_frames": 209, - "fps": 24, - "weight": 1 - } - ], - "extra-body": { + "dataset_name": "random-mm", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "num_warmups": 1, + "random_input_len": 64, + "random_output_len": 1, + "random_range_ratio": 0.0, + "percentile_metrics": "e2el", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "video": 1 + }, + "random_mm_bucket_config": { + "(768, 1344, 209)": 1.0 + }, + "extra_body": { + "width": 1344, + "height": 768, + "num_inference_steps": 8, + "num_frames": 209, + "fps": 24, + "seed": 42, "flow_shift": 12.0, "extra_params": "{\"task\":\"ref2va\",\"duration\":8.7,\"aspect_ratio\":\"16:9\",\"audio_flow_shift\":3.0}" }, "baseline": { "H100": { - "throughput_qps": 0.0077, - "latency_mean": 105.7508, - "peak_memory_mb_mean": 55180.8 + "request_throughput": 0.0077, + "mean_e2el_ms": 105750.8, + "mean_peak_memory_mb": 55180.8 } } } @@ -219,7 +227,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 4 } @@ -228,8 +239,6 @@ "diffusion" ], "description": "MiniMax-H3 T2VA with 4-card distributed layerwise offload (DLO), 8 denoise steps (USP4 + HSDP4, encoder-TP4, VAE-PP4 tile)", - "server_type": "vllm-omni", - "benchmark_endpoint": "/v1/videos", "server_params": { "model": "MiniMaxAI/MiniMax-H3", "serve_args": { @@ -256,33 +265,29 @@ "benchmark_params": [ { "name": "1344x768_frames209_steps8", - "dataset": "random", - "task": "t2v", - "num-prompts": 3, - "max-concurrency": 1, - "warmup-requests": 1, - "warmup-concurrency": 1, - "seed": 42, - "random-request-config": [ - { - "width": 1344, - "height": 768, - "prompt": "Cinematic drone shot pushing through misty mountain ridges at sunrise, golden light catching the clouds, slow and smooth.", - "num_inference_steps": 8, - "num_frames": 209, - "fps": 24, - "weight": 1 - } - ], - "extra-body": { + "dataset_name": "random", + "endpoint": "/v1/videos", + "num_prompts": 3, + "max_concurrency": 1, + "num_warmups": 1, + "random_input_len": 64, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 1344, + "height": 768, + "num_inference_steps": 8, + "num_frames": 209, + "fps": 24, + "seed": 42, "flow_shift": 12.0, "extra_params": "{\"task\":\"t2va\",\"duration\":8.7,\"aspect_ratio\":\"16:9\",\"audio_flow_shift\":3.0}" }, "baseline": { "H100": { - "throughput_qps": 0.0212, - "latency_mean": 38.2089, - "peak_memory_mb_mean": 36218.4 + "request_throughput": 0.0212, + "mean_e2el_ms": 38208.9, + "mean_peak_memory_mb": 36218.4 } } } diff --git a/tests/dfx/perf/tests/test_runner_metadata.py b/tests/dfx/perf/tests/test_runner_metadata.py index df99f090dcb..f7ce0550a8a 100644 --- a/tests/dfx/perf/tests/test_runner_metadata.py +++ b/tests/dfx/perf/tests/test_runner_metadata.py @@ -205,15 +205,53 @@ def test_is_diffusion_perf_config(): from tests.dfx.conftest import is_diffusion_perf_config assert not is_diffusion_perf_config( - {"test_name": "omni_a", "mark": [{"hardware_marks": {"res": {"cuda": "H100"}}}, "omni"]} + { + "test_name": "omni_a", + "mark": [{"hardware_marks": {"res": {"cuda": "H100"}}}, "omni"], + "benchmark_params": [{"dataset_name": "random", "endpoint": "/v1/chat/completions"}], + } ) assert is_diffusion_perf_config( { "test_name": "diff_a", "server_type": "vllm-omni", "mark": [{"hardware_marks": {"res": {"cuda": "H100"}}}, "diffusion"], + "benchmark_params": [{"task": "t2i", "dataset": "random"}], } ) + videos_cfg = { + "test_name": "diff_videos", + "server_type": "vllm-omni", + "mark": [{"hardware_marks": {"res": {"cuda": "H100"}}}, "diffusion"], + "benchmark_params": [{"task": "t2v", "dataset_name": "random", "endpoint": "/v1/videos"}], + } + assert not is_diffusion_perf_config(videos_cfg) + custom_edits_cfg = { + "test_name": "diff_custom_edits", + "server_type": "vllm-omni", + "benchmark_endpoint": "/v1/images/edits", + "benchmark_params": [{"dataset": "custom", "task": "ti2i"}], + } + assert is_diffusion_perf_config(custom_edits_cfg) + + +def test_merge_omni_default_server_args_respects_json(): + from tests.dfx.perf.scripts.run_benchmark import _merge_omni_default_server_args + + extra = ("--stage-init-timeout", "1800", "--init-timeout", "1800", "--usp", "4") + assert _merge_omni_default_server_args(extra, use_omni=True) == [] + assert _merge_omni_default_server_args((), use_omni=True) == [ + "--stage-init-timeout", + "600", + "--init-timeout", + "900", + ] + assert _merge_omni_default_server_args((), use_omni=False) == [] + # Only fill the missing default; keep JSON's other timeouts. + assert _merge_omni_default_server_args(("--stage-init-timeout=1800",), use_omni=True) == [ + "--init-timeout", + "900", + ] def test_benchmark_param_id_suffix_from_task_eval_phase(): diff --git a/tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json b/tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json index 3244d1c4155..7c288ffb168 100644 --- a/tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json +++ b/tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json @@ -5,7 +5,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 1 } @@ -14,7 +17,6 @@ "diffusion" ], "description": "Single-device baseline", - "server_type": "vllm-omni", "server_params": { "model": "Wan-AI/Wan2.2-I2V-A14B-Diffusers", "serve_args": { @@ -24,28 +26,36 @@ "benchmark_params": [ { "name": "832x480_frames81_steps4", - "dataset": "random", - "task": "i2v", - "num-prompts": 10, - "max-concurrency": 1, - "num-input-images": 1, - "seed": 42, - "enable-negative-prompt": true, - "random-request-config": [ - { - "width": 832, - "height": 480, - "num_inference_steps": 4, - "num_frames": 81, - "fps": 16, - "weight": 1 - } - ], + "dataset_name": "random-mm", + "endpoint": "/v1/videos", + "num_prompts": 10, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "random_range_ratio": 0.0, + "percentile_metrics": "e2el", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(480, 832, 1)": 1.0 + }, + "extra_body": { + "width": 832, + "height": 480, + "num_inference_steps": 4, + "num_frames": 81, + "fps": 16, + "seed": 42, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, "baseline": { "H100": { - "throughput_qps": 0.0361, - "latency_mean": 27.6206, - "peak_memory_mb_mean": 80548.0 + "request_throughput": 0.0361, + "mean_e2el_ms": 27620.6, + "mean_peak_memory_mb": 80548.0 } } } @@ -57,7 +67,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"], + "cuda": [ + "H100", + "B200" + ], "npu": "A3" }, "num_cards": 2 @@ -67,7 +80,6 @@ "diffusion" ], "description": "USP=2 + VAE patch parallel=2 + HSDP + VAE slicing", - "server_type": "vllm-omni", "server_params": { "model": "Wan-AI/Wan2.2-I2V-A14B-Diffusers", "serve_args": { @@ -81,55 +93,71 @@ "benchmark_params": [ { "name": "832x480_frames81_steps4", - "dataset": "random", - "task": "i2v", - "num-prompts": 10, - "max-concurrency": 1, - "num-input-images": 1, - "seed": 42, - "enable-negative-prompt": true, - "random-request-config": [ - { - "width": 832, - "height": 480, - "num_inference_steps": 4, - "num_frames": 81, - "fps": 16, - "weight": 1 - } - ], + "dataset_name": "random-mm", + "endpoint": "/v1/videos", + "num_prompts": 10, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "random_range_ratio": 0.0, + "percentile_metrics": "e2el", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(480, 832, 1)": 1.0 + }, + "extra_body": { + "width": 832, + "height": 480, + "num_inference_steps": 4, + "num_frames": 81, + "fps": 16, + "seed": 42, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, "baseline": { "H100": { - "throughput_qps": 0.05263, - "latency_mean": 18.9415, - "peak_memory_mb_mean": 50053.7 + "request_throughput": 0.05263, + "mean_e2el_ms": 18941.5, + "mean_peak_memory_mb": 50053.7 } } }, { "name": "1280x720_frames121_steps4", - "dataset": "random", - "task": "i2v", - "num-prompts": 10, - "max-concurrency": 1, - "num-input-images": 1, - "seed": 42, - "enable-negative-prompt": true, - "random-request-config": [ - { - "width": 1280, - "height": 720, - "num_inference_steps": 4, - "num_frames": 121, - "fps": 16, - "weight": 1 - } - ], + "dataset_name": "random-mm", + "endpoint": "/v1/videos", + "num_prompts": 10, + "max_concurrency": 1, + "random_input_len": 64, + "random_output_len": 1, + "random_range_ratio": 0.0, + "percentile_metrics": "e2el", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(720, 1280, 1)": 1.0 + }, + "extra_body": { + "width": 1280, + "height": 720, + "num_inference_steps": 4, + "num_frames": 121, + "fps": 16, + "seed": 42, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, "baseline": { "H100": { - "throughput_qps": 0.009595, - "latency_mean": 104.1058, - "peak_memory_mb_mean": 59718.0 + "request_throughput": 0.009595, + "mean_e2el_ms": 104105.8, + "mean_peak_memory_mb": 59718.0 } } } diff --git a/tools/nightly/run_nightly_jobs.sh b/tools/nightly/run_nightly_jobs.sh index c1cca73e2fa..d0e9c2daa58 100644 --- a/tools/nightly/run_nightly_jobs.sh +++ b/tools/nightly/run_nightly_jobs.sh @@ -41,7 +41,7 @@ # From repo root: pytest -sv -m " and local_model" (markers from MODEL_TYPE: omni, tts, # diffusion; all → "(omni or tts or diffusion) and local_model"). Not filtered by nightly YAML. # LABEL_SUBSTR: if set, restrict to tests/**/test_*.py basenames and tests/dfx/perf/tests/*.json -# whose filename contains the substring (benchmark runner chosen by JSON family). +# whose filename contains the substring (benchmark runner via is_diffusion_perf_config). # # stability (when included in TEST_TYPE): # From repo root: pytest -s -v --run-level full_model -m "" tests/dfx/stability/scripts/... @@ -686,8 +686,19 @@ def perf_json_model_family(json_basename: str) -> str: def perf_json_runner(json_basename: str) -> Path: - if perf_json_model_family(json_basename) in ("omni", "tts"): + """Pick benchmark runner from JSON schema (``dataset`` vs ``dataset_name``).""" + json_path = REPO_ROOT / PERF_TESTS_REL / json_basename + try: + if str(REPO_ROOT) not in sys.path: + sys.path.insert(0, str(REPO_ROOT)) + from tests.dfx.conftest import is_diffusion_perf_config, load_configs + + configs = load_configs(str(json_path)) + if configs and any(is_diffusion_perf_config(cfg) for cfg in configs): + return RUN_DIFFUSION_BENCHMARK_REL return RUN_BENCHMARK_REL + except Exception as exc: + print(f"# warn: could not classify {json_basename}: {exc}", file=sys.stderr) return RUN_DIFFUSION_BENCHMARK_REL From 7c92ee793e7a6612909a6dfb3d077ef9706b4018 Mon Sep 17 00:00:00 2001 From: wangyu <410167048@qq.com> Date: Thu, 17 Sep 2026 19:40:17 +0800 Subject: [PATCH 03/15] Add metrics aggregation and printing for stage durations in benchmarks - Introduced `aggregate_stage_durations` and `print_stage_durations_metrics` functions to compute and display mean, p50, and p99 stage durations from request outputs. - Updated `MixRequestFuncOutput` to include `stage_durations` for tracking per-stage timings. - Enhanced test coverage with new tests for aggregating and printing stage durations metrics. - Refactored existing code to utilize `SimpleNamespace` for cleaner tokenization output. This update improves the observability of performance metrics during benchmarking, facilitating better analysis of stage timings. Signed-off-by: [Your Name] Signed-off-by: wangyu <410167048@qq.com> --- tests/benchmarks/metrics/test_metrics.py | 43 +++++++-- tests/benchmarks/patch/test_patch.py | 71 +++++++++++++- vllm_omni/benchmarks/metrics/metrics.py | 70 ++++++++++++-- vllm_omni/benchmarks/patch/patch.py | 115 ++++++++++++++++++++--- 4 files changed, 266 insertions(+), 33 deletions(-) diff --git a/tests/benchmarks/metrics/test_metrics.py b/tests/benchmarks/metrics/test_metrics.py index 0aae841afa4..bb080ee2d00 100644 --- a/tests/benchmarks/metrics/test_metrics.py +++ b/tests/benchmarks/metrics/test_metrics.py @@ -6,11 +6,16 @@ """ import math +from types import SimpleNamespace import pytest from vllm.benchmarks.serve import TaskType -from vllm_omni.benchmarks.metrics.metrics import calculate_metrics +from vllm_omni.benchmarks.metrics.metrics import ( + aggregate_stage_durations, + calculate_metrics, + print_stage_durations_metrics, +) from vllm_omni.benchmarks.patch.patch import MixRequestFuncOutput pytestmark = [pytest.mark.core_model, pytest.mark.benchmark, pytest.mark.cpu] @@ -379,12 +384,7 @@ class _EmptyAwareTokenizer: """ def __call__(self, text, add_special_tokens=False): - class _R: - pass - - r = _R() - r.input_ids = [0] * len(text) - return r + return SimpleNamespace(input_ids=[0] * len(text)) def _make_tts_output(prompt_len: int) -> MixRequestFuncOutput: @@ -588,5 +588,34 @@ def test_image_with_generated_text_still_reports_text_result(capsys): assert " Image Result " in out +def test_aggregate_stage_durations_mean_p50_p99() -> None: + ok_a = MixRequestFuncOutput() + ok_a.success = True + ok_a.stage_durations = {"diffuse": 1.0, "vae.decode": 0.2} + ok_b = MixRequestFuncOutput() + ok_b.success = True + ok_b.stage_durations = {"diffuse": 3.0, "vae.decode": 0.4} + failed = MixRequestFuncOutput() + failed.success = False + failed.stage_durations = {"diffuse": 99.0} + + summaries = aggregate_stage_durations([ok_a, ok_b, failed]) + assert summaries["stage_durations_mean"]["diffuse"] == pytest.approx(2.0) + assert summaries["stage_durations_p50"]["diffuse"] == pytest.approx(2.0) + assert summaries["stage_durations_mean"]["vae.decode"] == pytest.approx(0.3) + assert "stage_durations_p99" in summaries + + +def test_print_stage_durations_metrics(capsys) -> None: + output = MixRequestFuncOutput() + output.success = True + output.stage_durations = {"diffuse": 1.25, "text_encoder.forward": 0.5} + print_stage_durations_metrics([output]) + out = capsys.readouterr().out + assert "Stage Durations Mean (s):" in out + assert "diffuse" in out + assert "text_encoder.forward" in out + + if __name__ == "__main__": pytest.main([__file__, "-v", "-s"]) diff --git a/tests/benchmarks/patch/test_patch.py b/tests/benchmarks/patch/test_patch.py index 60e071ad5d9..98f5843d5d9 100644 --- a/tests/benchmarks/patch/test_patch.py +++ b/tests/benchmarks/patch/test_patch.py @@ -24,10 +24,12 @@ MixRequestFuncOutput, _add_video_extra_body_to_form, _add_video_reference_to_form, + _apply_image_metrics_from_payload, _apply_stage0_token_timings, _apply_video_metrics_from_payload, _attach_seed_tts_to_request_func_input, _build_benchmark_session, + _extract_stage_durations_from_payload, _omni_request_timeout_s, async_request_openai_chat_omni_completions, async_request_openai_image_edits_omni, @@ -134,7 +136,7 @@ async def __aexit__(self, exc_type, exc_val, exc_tb): @pytest.mark.asyncio async def test_seed_tts_realtime_duplex_exports_per_request_metrics(monkeypatch): class FakeRealtimeClient: - last_instance = None + last_instance: "FakeRealtimeClient | None" = None def __init__(self, url): assert url == "ws://localhost:8000/v1/realtime?duplex=1" @@ -240,6 +242,8 @@ async def close_session(self, **_kwargs): ) client = FakeRealtimeClient.last_instance + assert client is not None + assert client.configure_kwargs is not None assert client.configure_kwargs["native_duplex"] is False assert client.configure_kwargs["extra_body"] == { "ref_audio": "data:audio/wav;base64,AAAA", @@ -1345,16 +1349,15 @@ async def __aexit__(self, *_args): return None captured_stream: list[str] = [] - real_add_field = None + import aiohttp + + real_add_field = aiohttp.FormData.add_field def tracking_add_field(self, name, value=None, **kwargs): if name == "stream": captured_stream.append(str(value)) return real_add_field(self, name, value, **kwargs) - import aiohttp - - real_add_field = aiohttp.FormData.add_field mocker.patch.object(aiohttp.FormData, "add_field", tracking_add_field) mock_session = mocker.AsyncMock() @@ -1527,6 +1530,7 @@ def tracking_add_field(self, name, value=None, **kwargs): assert field_names.count("image_reference") == 1 assert "input_reference" not in field_names payload = next(value for name, value in captured if name == "image_reference") + assert isinstance(payload, (str, bytes, bytearray)) assert json.loads(payload) == reference @@ -1540,5 +1544,62 @@ def test_video_unsupported_image_reference_raises() -> None: _add_video_reference_to_form(form, "/tmp/does-not-exist-ref.png") +def test_extract_stage_durations_from_video_and_image_shapes() -> None: + video_payload = { + "stage_durations": { + "diffuse": 1.5, + "text_encoder.forward": 0.2, + "vae.decode": 0.1, + "stage_0_gen_ms": 4000.0, + } + } + assert _extract_stage_durations_from_payload(video_payload) == video_payload["stage_durations"] + + image_stage_durations = { + "diffuse": 2.0, + "vae.decode": 0.3, + } + image_payload = { + "metrics": {"stage_durations": image_stage_durations}, + "data": [{"b64_json": "x"}], + } + assert _extract_stage_durations_from_payload(image_payload) == image_stage_durations + + +def test_video_metrics_persist_full_stage_durations() -> None: + output = MixRequestFuncOutput() + output.latency = 4.2 + stage_durations = { + "diffuse": 1.5, + "vae.decode": 0.25, + "stage_0_gen_ms": 4000.0, + } + payload = { + "duration_s": 2.0, + "num_frames": 48, + "fps": 24.0, + "stage_durations": stage_durations, + } + _apply_video_metrics_from_payload(output, payload, {}) + assert output.stage_durations == stage_durations + assert output.video_generation_time_ms == pytest.approx(4000.0) + + +def test_image_metrics_persist_stage_durations_from_metrics() -> None: + output = MixRequestFuncOutput() + stage_durations = { + "diffuse": 1.1, + "text_encoder.forward": 0.4, + "vae.decode": 0.2, + } + payload = { + "created": 1, + "data": [{"b64_json": _MIN_PNG_B64}], + "metrics": {"stage_durations": stage_durations}, + } + assert _apply_image_metrics_from_payload(output, payload) == 1 + assert output.stage_durations == stage_durations + + if __name__ == "__main__": pytest.main([__file__, "-v", "-s"]) diff --git a/vllm_omni/benchmarks/metrics/metrics.py b/vllm_omni/benchmarks/metrics/metrics.py index 759e606671f..13d730f920c 100644 --- a/vllm_omni/benchmarks/metrics/metrics.py +++ b/vllm_omni/benchmarks/metrics/metrics.py @@ -5,6 +5,7 @@ from collections import defaultdict from collections.abc import Sequence from dataclasses import field, make_dataclass +from typing import Any import numpy as np from vllm.benchmarks.datasets import SampleRequest @@ -69,7 +70,9 @@ (defs.PERCENTILES_PEAK_MEMORY_MB, _PERCENTILE_ROWS_TYPE, field(default=None)), ] -MultiModalsBenchmarkMetrics = make_dataclass( +# ``make_dataclass`` returns a runtime class that mypy treats as a variable, not +# a type. Annotate as Any so call sites can use these names in annotations. +MultiModalsBenchmarkMetrics: Any = make_dataclass( "MultiModalsBenchmarkMetrics", _MULTIMODAL_BENCHMARK_FIELDS, bases=(BenchmarkMetrics,), @@ -99,7 +102,7 @@ (defs.INTER_OUTPUT_LATENCIES_MS, _FLOAT_LIST_TYPE, field(default_factory=list)), ] -StageBenchmarkMetrics = make_dataclass( +StageBenchmarkMetrics: Any = make_dataclass( "StageBenchmarkMetrics", _STAGE_BENCHMARK_FIELDS, namespace={"__doc__": "Aggregated metrics for one pipeline stage (for printing only)."}, @@ -250,6 +253,7 @@ def print_metrics( print_image_metrics(selected_percentiles or [], metrics) if _has_video_output(metrics): print_video_metrics(selected_percentiles or [], metrics) + print_stage_durations_metrics(outputs) if print_stage and outputs and selected_percentiles is not None: stage_metrics = _build_stage_metrics_from_outputs(outputs) if stage_metrics: @@ -338,6 +342,50 @@ def print_peak_memory_metrics(metrics: MultiModalsBenchmarkMetrics): print("{:<40} {:<10.2f}".format(f"P{_p_label(p)} PEAK_MEMORY_MB (MB):", value)) +def aggregate_stage_durations(outputs: Sequence[RequestFuncOutput]) -> dict[str, dict[str, float]]: + """Aggregate per-request pipeline profiler timings into mean/p50/p99 maps. + + Mirrors ``diffusion_benchmark_serving`` so ``--save-result`` JSON can carry + ``stage_durations_{mean,p50,p99}`` for Diffuse / VAE / TextEncoder keys. + """ + stage_duration_lists: dict[str, list[float]] = {} + for output in outputs: + if not getattr(output, "success", False): + continue + stage_durations = getattr(output, "stage_durations", None) + if not isinstance(stage_durations, dict): + continue + for stage, duration in stage_durations.items(): + if isinstance(duration, bool) or not isinstance(duration, (int, float)) or not np.isfinite(duration): + continue + stage_duration_lists.setdefault(str(stage), []).append(float(duration)) + if not stage_duration_lists: + return {} + return { + "stage_durations_mean": {stage: float(np.mean(values)) for stage, values in stage_duration_lists.items()}, + "stage_durations_p50": { + stage: float(np.percentile(values, 50)) for stage, values in stage_duration_lists.items() + }, + "stage_durations_p99": { + stage: float(np.percentile(values, 99)) for stage, values in stage_duration_lists.items() + }, + } + + +def print_stage_durations_metrics(outputs: Sequence[RequestFuncOutput] | None) -> None: + """Print mean pipeline profiler stage durations when any request reported them.""" + if not outputs: + return + summaries = aggregate_stage_durations(outputs) + mean = summaries.get("stage_durations_mean") or {} + if not mean: + return + print("{s:{c}^{n}}".format(s=" Stage Durations ", n=50, c="-")) + print("Stage Durations Mean (s):") + for stage, value in mean.items(): + print("{:<40} {:<10.4f}".format(f" {stage}:", value)) + + def print_image_metrics(selected_percentiles: list[float], metrics: MultiModalsBenchmarkMetrics): print("{s:{c}^{n}}".format(s=" Image Result ", n=50, c="=")) print("{:<40} {:<10}".format("Total images generated:", getattr(metrics, defs.TOTAL_IMAGES))) @@ -818,7 +866,7 @@ def calculate_metrics( request_rate, benchmark_duration, print_stage: bool = False, -) -> tuple[BenchmarkMetrics, list[int]]: +) -> tuple[MultiModalsBenchmarkMetrics, list[int]]: """Calculate the metrics for the benchmark. Args: @@ -928,7 +976,7 @@ def calculate_metrics( goodput_ttfts.append(outputs[i].ttft if ttft_measured else None) goodput_audio_ttfps.append(getattr(outputs[i], defs.AUDIO_TTFP, 0.0) if audio_ttfp_measured else None) audio_duration.append(getattr(outputs[i], defs.AUDIO_DURATION, 0.0)) - audio_frames.append(getattr(outputs[i], defs.AUDIO_FRAMES, 0.0)) + audio_frames.append(getattr(outputs[i], defs.AUDIO_FRAMES, 0)) image_count = int(getattr(outputs[i], defs.IMAGE_COUNT, 0) or 0) total_images += image_count image_generation_time_ms = float(getattr(outputs[i], defs.IMAGE_GENERATION_TIME_MS, 0.0) or 0.0) @@ -982,9 +1030,17 @@ def calculate_metrics( good_completed += 1 if completed == 0: - warnings.formatwarning = lambda msg, category, filename, lineno, line=None: ( - f"{filename}:{lineno}: {category.__name__}: {msg}\n" - ) + + def _formatwarning( + message: Warning | str, + category: type[Warning], + filename: str, + lineno: int, + line: str | None = None, + ) -> str: + return f"{filename}:{lineno}: {category.__name__}: {message}\n" + + warnings.formatwarning = _formatwarning warnings.warn( "All requests failed. This is likely due to a misconfiguration on the benchmark arguments.", stacklevel=2, diff --git a/vllm_omni/benchmarks/patch/patch.py b/vllm_omni/benchmarks/patch/patch.py index a230b9c2e12..6f608de8050 100644 --- a/vllm_omni/benchmarks/patch/patch.py +++ b/vllm_omni/benchmarks/patch/patch.py @@ -14,7 +14,7 @@ import traceback import uuid import wave -from collections.abc import Iterable, Mapping +from collections.abc import Iterable, Iterator, Mapping from dataclasses import dataclass, replace from datetime import datetime from pathlib import Path @@ -404,9 +404,13 @@ def _prepare_omniinteract_batch(input_requests: list[SampleRequest]) -> None: for sample in input_requests: if not isinstance(sample, OmniInteractSampleRequest): continue - root = sample.omniinteract_options.output_root.resolve() + options = sample.omniinteract_options + case = sample.omniinteract_case + if options is None or case is None: + continue + root = options.output_root.resolve() roots.add(root) - clear_case_artifacts(sample.omniinteract_options.output_root, sample.omniinteract_case) + clear_case_artifacts(options.output_root, case) for root in roots: clear_batch_artifacts(root) @@ -711,7 +715,7 @@ def get_samples(args, tokenizer): _serve_mod = sys.modules.get("vllm.benchmarks.serve") if _serve_mod is not None: - _serve_mod.get_samples = get_samples + setattr(_serve_mod, "get_samples", get_samples) @dataclass @@ -748,6 +752,9 @@ class MixRequestFuncOutput(RequestFuncOutput): tts_turn_pcm_bytes: list[bytes] | None = None #: Per-stage snapshot from orchestrator ``metrics["stage_metrics"]`` (merged across SSE chunks). stage_metrics: dict[str, dict] | None = None + #: Diffusion pipeline profiler timings from response ``stage_durations`` + #: (e.g. diffuse / text_encoder.forward / vae.decode), when present. + stage_durations: dict[str, float] | None = None stage_id: int | None = None final_output_type: str | None = None duplex_request_metrics: list[dict[str, object]] | None = None @@ -1012,6 +1019,68 @@ def _update_output_peak_memory_from_payload(output: MixRequestFuncOutput, data: output.peak_memory_mb = peak_memory_mb +def _coerce_stage_durations_dict(raw: object) -> dict[str, float] | None: + """Normalize a stage_durations mapping to ``dict[str, float]``.""" + if not isinstance(raw, dict) or not raw: + return None + coerced: dict[str, float] = {} + for key, value in raw.items(): + if isinstance(value, bool) or not isinstance(value, (int, float)) or not np.isfinite(value): + continue + coerced[str(key)] = float(value) + return coerced or None + + +def _extract_stage_durations_from_payload(data: Mapping[str, object]) -> dict[str, float] | None: + """Pull pipeline profiler timings from video/image/chat response shapes.""" + found = _coerce_stage_durations_dict(data.get("stage_durations")) + if found: + return found + + metrics = data.get("metrics") + if isinstance(metrics, dict): + found = _coerce_stage_durations_dict(metrics.get("stage_durations")) + if found: + return found + + response_data = data.get("data") + if isinstance(response_data, list): + for item in response_data: + if isinstance(item, dict): + found = _coerce_stage_durations_dict(item.get("stage_durations")) + if found: + return found + + choices = data.get("choices") + if isinstance(choices, list): + for choice in choices: + if not isinstance(choice, dict): + continue + for message_key in ("message", "delta"): + message = choice.get(message_key) + if not isinstance(message, dict): + continue + content = message.get("content") + if isinstance(content, list): + for item in content: + if isinstance(item, dict): + found = _coerce_stage_durations_dict(item.get("stage_durations")) + if found: + return found + elif isinstance(content, dict): + found = _coerce_stage_durations_dict(content.get("stage_durations")) + if found: + return found + return None + + +def _update_output_stage_durations_from_payload(output: MixRequestFuncOutput, data: Mapping[str, object]) -> None: + """Persist the full profiler ``stage_durations`` map when the response has one.""" + found = _extract_stage_durations_from_payload(data) + if found: + output.stage_durations = found + + def _image_metrics_from_stage_metrics(metrics: object) -> tuple[int, float, int, float]: if not isinstance(metrics, dict): return 0, 0.0, 0, 0.0 @@ -1084,6 +1153,7 @@ def _apply_image_metrics_from_payload(output: MixRequestFuncOutput, data: Mappin """Populate image benchmark fields from an OpenAI-compatible image payload.""" _update_output_stage_metrics_from_payload(output, data, update_output_tokens=False) _update_output_peak_memory_from_payload(output, data) + _update_output_stage_durations_from_payload(output, data) payload_image_count = 0 response_data = data.get("data") @@ -1296,8 +1366,9 @@ def _apply_video_metrics_from_payload( output.video_frames = _video_frames_from_payload(data, request_body) _update_output_stage_metrics_from_payload(output, data, update_output_tokens=False) _update_output_peak_memory_from_payload(output, data) + _update_output_stage_durations_from_payload(output, data) - stage_durations = data.get("stage_durations") + stage_durations = output.stage_durations if output.stage_durations is not None else data.get("stage_durations") stage_gen_ms = _video_generation_ms_from_stage_durations(stage_durations) if stage_gen_ms <= 0: inference_time_s = coerce_positive_float_scalar(data.get("inference_time_s")) @@ -1409,6 +1480,7 @@ async def async_request_openai_chat_omni_completions( output.image_pixels = 0 output.denoise_step_latency_ms = 0.0 output.peak_memory_mb = 0.0 + output.stage_durations = None completion_tokens_seen = 0 try: async with session.post(url=api_url, json=payload, headers=headers) as response: @@ -1439,6 +1511,7 @@ async def async_request_openai_chat_omni_completions( data = json.loads(chunk) _update_output_stage_metrics_from_payload(output, data) _update_output_peak_memory_from_payload(output, data) + _update_output_stage_durations_from_payload(output, data) usage = data.get("usage") completion_tokens = None if isinstance(usage, dict): @@ -1977,6 +2050,7 @@ async def async_request_openai_image_edits_omni( update_output_tokens=(data.get("type") == "ar_delta"), ) _update_output_peak_memory_from_payload(output, data) + _update_output_stage_durations_from_payload(output, data) chunk_type = data.get("type") if chunk_type == "ar_delta": @@ -2531,9 +2605,12 @@ async def async_request_openai_realtime_duplex( output.ttft = (metric_mean(session_metrics.get("ttft_ms")) or 0.0) / 1000.0 output.audio_ttfp = (metric_mean(session_metrics.get("ttfp_ms")) or 0.0) / 1000.0 output.audio_rtf = metric_mean(session_metrics.get("rtf")) or 0.0 - output.audio_duration = ( - sum(float(metric.get("audio_duration_ms") or 0.0) for metric in turn_metrics) / 1000.0 - ) + audio_duration_ms = 0.0 + for metric in turn_metrics: + raw_duration = metric.get("audio_duration_ms") or 0.0 + if isinstance(raw_duration, (int, float)) and not isinstance(raw_duration, bool): + audio_duration_ms += float(raw_duration) + output.audio_duration = audio_duration_ms / 1000.0 output.audio_frames = int(output.audio_duration * client.events.output_sample_rate_hz) output.latency = request_finished_at - output.start_time output.tts_turn_pcm_bytes = turn_pcm_bytes @@ -2614,6 +2691,7 @@ async def async_request_openai_realtime_duplex( from vllm_omni.benchmarks.metrics.metrics import ( MultiModalsBenchmarkMetrics, + aggregate_stage_durations, calculate_metrics, has_metric_samples, ) @@ -2732,7 +2810,7 @@ async def benchmark( if num_warmups > 0: print(f"Warming up with {num_warmups} requests...") warmup_pbar = None if disable_tqdm else tqdm(total=num_warmups) - warmup_semaphore = asyncio.Semaphore(max_concurrency) if max_concurrency else contextlib.nullcontext() + warmup_semaphore: Any = asyncio.Semaphore(max_concurrency) if max_concurrency else contextlib.nullcontext() warmup_tasks = [] async def warmup_limited_request_func(): @@ -2753,9 +2831,13 @@ async def warmup_limited_request_func(): if lora_modules: lora_modules_list = list(lora_modules) if lora_assignment == "round-robin": - lora_modules = iter([lora_modules_list[i % len(lora_modules_list)] for i in range(len(input_requests))]) + lora_module_iter: Iterator[str] | None = iter( + [lora_modules_list[i % len(lora_modules_list)] for i in range(len(input_requests))] + ) else: - lora_modules = iter([random.choice(lora_modules_list) for _ in range(len(input_requests))]) + lora_module_iter = iter([random.choice(lora_modules_list) for _ in range(len(input_requests))]) + else: + lora_module_iter = None if profile: print("Starting profiler...") @@ -2794,7 +2876,7 @@ async def warmup_limited_request_func(): pbar = None if disable_tqdm else tqdm(total=len(input_requests)) - semaphore = asyncio.Semaphore(max_concurrency) if max_concurrency else contextlib.nullcontext() + semaphore: Any = asyncio.Semaphore(max_concurrency) if max_concurrency else contextlib.nullcontext() async def limited_request_func(request_func_input, session, pbar): async with semaphore: @@ -2865,8 +2947,8 @@ async def probe_loop(): ) per_request_extra_body = _merge_overrides(extra_body, request.request_overrides) req_model_id, req_model_name = model_id, model_name - if lora_modules: - req_lora_module = next(lora_modules) + if lora_module_iter is not None: + req_lora_module = next(lora_module_iter) req_model_id, req_model_name = req_lora_module, req_lora_module request_func_input = RequestFuncInput( @@ -2903,6 +2985,7 @@ async def probe_loop(): omniinteract_summary = _finalize_omniinteract_batch(input_requests, outputs) + actual_output_lens: list[int] | int if task_type == TaskType.GENERATION: metrics, actual_output_lens = calculate_metrics( input_requests=input_requests, @@ -3023,6 +3106,10 @@ def measured_ttft(output: RequestFuncOutput) -> float | None: if omniinteract_summary is not None: result["omniinteract"] = omniinteract_summary + stage_duration_summaries = aggregate_stage_durations(outputs) + if stage_duration_summaries: + result.update(stage_duration_summaries) + from vllm_omni.benchmarks.data_modules.daily_omni_eval import ( compute_daily_omni_accuracy_metrics, print_daily_omni_accuracy_summary, From 5c3ca5862d13c93db9530bb6e4ea09e42131c35c Mon Sep 17 00:00:00 2001 From: wangyu <410167048@qq.com> Date: Thu, 17 Sep 2026 22:39:28 +0800 Subject: [PATCH 04/15] Update performance test configurations to use GPT-2 tokenizer and reduce random input length - Added "tokenizer": "gpt2" to multiple video performance test configurations for consistency. - Reduced "random_input_len" from 64 to 8 across various test cases to optimize input handling. This change enhances the uniformity of tokenizer usage and improves the efficiency of input processing in performance tests. Signed-off-by: wangyu <410167048@qq.com> --- .../perf/tests/test_cosmos3_vllm_omni.json | 12 +++++--- .../test_hunyuanvideo15_i2v_vllm_omni.json | 6 ++-- .../test_hunyuanvideo15_t2v_vllm_omni.json | 6 ++-- .../tests/test_lingbot_video_vllm_omni.json | 3 +- tests/dfx/perf/tests/test_ltx2_vllm_omni.json | 30 ++++++++++++------- .../perf/tests/test_minimax_h3_vllm_omni.json | 12 +++++--- .../perf/tests/test_wan22_i2v_vllm_omni.json | 9 ++++-- 7 files changed, 52 insertions(+), 26 deletions(-) diff --git a/tests/dfx/perf/tests/test_cosmos3_vllm_omni.json b/tests/dfx/perf/tests/test_cosmos3_vllm_omni.json index 06bf70d50c7..662f845f648 100644 --- a/tests/dfx/perf/tests/test_cosmos3_vllm_omni.json +++ b/tests/dfx/perf/tests/test_cosmos3_vllm_omni.json @@ -36,10 +36,11 @@ { "name": "1024x1024_steps4", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/images/generations", "num_prompts": 3, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { @@ -98,10 +99,11 @@ { "name": "1280x720_frames189_steps4", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { @@ -165,10 +167,11 @@ { "name": "1280x720_frames189_steps4", "dataset_name": "random-mm", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "random_range_ratio": 0.0, "percentile_metrics": "e2el", @@ -239,10 +242,11 @@ { "name": "1280x720_frames189_steps4", "dataset_name": "random-mm", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "random_range_ratio": 0.0, "percentile_metrics": "e2el", diff --git a/tests/dfx/perf/tests/test_hunyuanvideo15_i2v_vllm_omni.json b/tests/dfx/perf/tests/test_hunyuanvideo15_i2v_vllm_omni.json index 925ccb23941..f14e77bcb92 100644 --- a/tests/dfx/perf/tests/test_hunyuanvideo15_i2v_vllm_omni.json +++ b/tests/dfx/perf/tests/test_hunyuanvideo15_i2v_vllm_omni.json @@ -28,10 +28,11 @@ { "name": "832x480_frames33_steps4", "dataset_name": "random-mm", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 10, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "random_range_ratio": 0.0, "percentile_metrics": "e2el", @@ -90,10 +91,11 @@ { "name": "832x480_frames33_steps4", "dataset_name": "random-mm", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 10, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "random_range_ratio": 0.0, "percentile_metrics": "e2el", diff --git a/tests/dfx/perf/tests/test_hunyuanvideo15_t2v_vllm_omni.json b/tests/dfx/perf/tests/test_hunyuanvideo15_t2v_vllm_omni.json index 65c2f85739a..021c5692ee1 100644 --- a/tests/dfx/perf/tests/test_hunyuanvideo15_t2v_vllm_omni.json +++ b/tests/dfx/perf/tests/test_hunyuanvideo15_t2v_vllm_omni.json @@ -28,10 +28,11 @@ { "name": "832x480_frames33_steps4", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 10, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { @@ -81,10 +82,11 @@ { "name": "832x480_frames33_steps4", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 10, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { diff --git a/tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json b/tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json index 88667e12f9e..9f85a14a2d5 100644 --- a/tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json +++ b/tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json @@ -28,11 +28,12 @@ { "name": "320x192_frames9_steps2", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, "num_warmups": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { diff --git a/tests/dfx/perf/tests/test_ltx2_vllm_omni.json b/tests/dfx/perf/tests/test_ltx2_vllm_omni.json index c5f986981fb..6b4c9eb702b 100644 --- a/tests/dfx/perf/tests/test_ltx2_vllm_omni.json +++ b/tests/dfx/perf/tests/test_ltx2_vllm_omni.json @@ -25,10 +25,11 @@ { "name": "256x256_145f_steps6", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { @@ -43,10 +44,11 @@ { "name": "480x768_41f_steps20", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { @@ -85,10 +87,11 @@ { "name": "256x256_145f_steps6", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { @@ -103,10 +106,11 @@ { "name": "480x768_41f_steps20", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { @@ -147,10 +151,11 @@ { "name": "256x256_145f_steps6", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { @@ -165,10 +170,11 @@ { "name": "480x768_41f_steps20", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { @@ -208,10 +214,11 @@ { "name": "256x256_145f_steps6", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { @@ -226,10 +233,11 @@ { "name": "480x768_41f_steps20", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { @@ -270,10 +278,11 @@ { "name": "256x256_145f_steps6", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { @@ -288,10 +297,11 @@ { "name": "480x768_41f_steps20", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { diff --git a/tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json b/tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json index dfd7d815f95..fb64945356d 100644 --- a/tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json +++ b/tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json @@ -40,11 +40,12 @@ { "name": "1344x768_frames209_steps8", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, "num_warmups": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { @@ -108,11 +109,12 @@ { "name": "1344x768_frames209_steps8", "dataset_name": "random-mm", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, "num_warmups": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "random_range_ratio": 0.0, "percentile_metrics": "e2el", @@ -185,11 +187,12 @@ { "name": "1344x768_frames209_steps8", "dataset_name": "random-mm", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, "num_warmups": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "random_range_ratio": 0.0, "percentile_metrics": "e2el", @@ -266,11 +269,12 @@ { "name": "1344x768_frames209_steps8", "dataset_name": "random", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 3, "max_concurrency": 1, "num_warmups": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "percentile_metrics": "e2el", "extra_body": { diff --git a/tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json b/tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json index 7c288ffb168..0aa060b4cbc 100644 --- a/tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json +++ b/tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json @@ -27,10 +27,11 @@ { "name": "832x480_frames81_steps4", "dataset_name": "random-mm", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 10, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "random_range_ratio": 0.0, "percentile_metrics": "e2el", @@ -94,10 +95,11 @@ { "name": "832x480_frames81_steps4", "dataset_name": "random-mm", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 10, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "random_range_ratio": 0.0, "percentile_metrics": "e2el", @@ -129,10 +131,11 @@ { "name": "1280x720_frames121_steps4", "dataset_name": "random-mm", + "tokenizer": "gpt2", "endpoint": "/v1/videos", "num_prompts": 10, "max_concurrency": 1, - "random_input_len": 64, + "random_input_len": 8, "random_output_len": 1, "random_range_ratio": 0.0, "percentile_metrics": "e2el", From 89751842e774bdbcf7c3cd552820e314877309be Mon Sep 17 00:00:00 2001 From: wangyu <410167048@qq.com> Date: Thu, 17 Sep 2026 23:04:10 +0800 Subject: [PATCH 05/15] Enhance stage duration metrics display in benchmarks - Updated `print_stage_durations_metrics` to format output with clearer labels and units for mean, median, and p99 stage durations. - Introduced a new helper function `_stage_duration_display_name` to prettify stage names for better readability in printed metrics. - Modified test cases to reflect changes in output formatting and ensure accurate assertions. This update improves the clarity and usability of performance metrics during benchmarking, aiding in the analysis of stage timings. Signed-off-by: wangyu <410167048@qq.com> --- tests/benchmarks/metrics/test_metrics.py | 15 +++++++++--- vllm_omni/benchmarks/metrics/metrics.py | 31 +++++++++++++++++++++--- 2 files changed, 38 insertions(+), 8 deletions(-) diff --git a/tests/benchmarks/metrics/test_metrics.py b/tests/benchmarks/metrics/test_metrics.py index bb080ee2d00..6e572e63a80 100644 --- a/tests/benchmarks/metrics/test_metrics.py +++ b/tests/benchmarks/metrics/test_metrics.py @@ -609,12 +609,19 @@ def test_aggregate_stage_durations_mean_p50_p99() -> None: def test_print_stage_durations_metrics(capsys) -> None: output = MixRequestFuncOutput() output.success = True - output.stage_durations = {"diffuse": 1.25, "text_encoder.forward": 0.5} + output.stage_durations = { + "Wan22I2VPipeline.diffuse": 1.25, + "Wan22I2VPipeline.text_encoder.forward": 0.4, + "queue_wait_ms": 0.5, + } print_stage_durations_metrics([output]) out = capsys.readouterr().out - assert "Stage Durations Mean (s):" in out - assert "diffuse" in out - assert "text_encoder.forward" in out + assert "Wan22I2VPipeline" not in out + assert "Mean Diffuse (s):" in out + assert "Median Diffuse (s):" in out + assert "P99 Diffuse (s):" in out + assert "Mean Text Encoder Forward (s):" in out + assert "Mean Queue Wait (ms):" in out if __name__ == "__main__": diff --git a/vllm_omni/benchmarks/metrics/metrics.py b/vllm_omni/benchmarks/metrics/metrics.py index 13d730f920c..f71cb291d1c 100644 --- a/vllm_omni/benchmarks/metrics/metrics.py +++ b/vllm_omni/benchmarks/metrics/metrics.py @@ -372,18 +372,41 @@ def aggregate_stage_durations(outputs: Sequence[RequestFuncOutput]) -> dict[str, } +def _stage_duration_display_name(stage: str) -> str: + """Strip pipeline-class prefix and prettify profiler keys for printing. + + ``Wan22I2VPipeline.diffuse`` -> ``Diffuse`` + ``Wan22I2VPipeline.text_encoder.forward`` -> ``Text Encoder Forward`` + ``queue_wait_ms`` / ``stage_0_gen_ms`` -> ``Queue Wait`` / ``Stage 0 Gen`` + """ + name = str(stage) + head, sep, tail = name.partition(".") + if sep and head.endswith("Pipeline"): + name = tail + if name.endswith("_ms"): + name = name[: -len("_ms")] + return name.replace("_", " ").replace(".", " ").title() + + def print_stage_durations_metrics(outputs: Sequence[RequestFuncOutput] | None) -> None: - """Print mean pipeline profiler stage durations when any request reported them.""" + """Print per-stage mean / median / p99 in the same style as other bench metrics.""" if not outputs: return summaries = aggregate_stage_durations(outputs) mean = summaries.get("stage_durations_mean") or {} if not mean: return + p50 = summaries.get("stage_durations_p50") or {} + p99 = summaries.get("stage_durations_p99") or {} print("{s:{c}^{n}}".format(s=" Stage Durations ", n=50, c="-")) - print("Stage Durations Mean (s):") - for stage, value in mean.items(): - print("{:<40} {:<10.4f}".format(f" {stage}:", value)) + for stage in mean: + label = _stage_duration_display_name(stage) + unit = " (ms)" if str(stage).endswith("_ms") else " (s)" + print("{:<40} {:<10.4f}".format(f"Mean {label}{unit}:", mean[stage])) + if stage in p50: + print("{:<40} {:<10.4f}".format(f"Median {label}{unit}:", p50[stage])) + if stage in p99: + print("{:<40} {:<10.4f}".format(f"P99 {label}{unit}:", p99[stage])) def print_image_metrics(selected_percentiles: list[float], metrics: MultiModalsBenchmarkMetrics): From ddc582b217c246e9d493d28f5c10238cf1f75285 Mon Sep 17 00:00:00 2001 From: wangyu <410167048@qq.com> Date: Thu, 17 Sep 2026 23:35:11 +0800 Subject: [PATCH 06/15] Refactor benchmark scripts to unify naming conventions and improve clarity - Updated references from `run_diffusion_benchmark.py` to `run_benchmark.py` across CI configurations and test files to standardize the script used for performance benchmarks. - Adjusted artifact paths and environment variable names for consistency in performance test configurations. - Enhanced documentation to clarify the distinction between diffusion and omni benchmarks, including updates to test examples and execution guides. This refactor aims to streamline the testing process and improve maintainability across the codebase. Signed-off-by: wangyu <410167048@qq.com> --- .../common/ci_source_file_dependencies.yml | 4 +- .buildkite/cuda/test-nightly.yml | 14 +- .buildkite/npu/test-npu-nightly.yml | 21 +- .../test_examples/l4_performance_tests.inc.md | 22 +- docs/contributing/ci/test_execution_guide.md | 4 +- docs/contributing/ci/test_writing_guide.md | 3 +- tests/dfx/perf/scripts/run_benchmark.py | 2 +- .../perf/scripts/run_diffusion_benchmark.py | 14 +- .../dfx/perf/tests/test_bagel_vllm_omni.json | 154 +++--- .../test_boogu_image_edit_vllm_omni.json | 481 ++++++++++-------- .../tests/test_boogu_image_vllm_omni.json | 169 +++--- .../tests/test_hunyuan_image_tp2_cfgp2.json | 31 +- .../tests/test_hunyuan_image_tp2_sp2.json | 31 +- .../perf/tests/test_hunyuan_image_tp4.json | 32 +- .../test_qwen_image_edit_2511_vllm_omni.json | 208 +++++--- .../test_qwen_image_layered_vllm_omni.json | 81 ++- .../perf/tests/test_qwen_image_vllm_omni.json | 301 +++++++---- 17 files changed, 943 insertions(+), 629 deletions(-) diff --git a/.buildkite/common/ci_source_file_dependencies.yml b/.buildkite/common/ci_source_file_dependencies.yml index 6795e9678b7..8ac6945360a 100644 --- a/.buildkite/common/ci_source_file_dependencies.yml +++ b/.buildkite/common/ci_source_file_dependencies.yml @@ -338,7 +338,7 @@ source_file_dependencies: diffusion_qwen_image_perf: - *qwen_image - - tests/dfx/perf/scripts/run_diffusion_benchmark.py + - tests/dfx/perf/scripts/run_benchmark.py - tests/dfx/perf/tests/test_qwen_image_vllm_omni.json diffusion_text_to_image_doc: @@ -370,7 +370,7 @@ source_file_dependencies: diffusion_hunyuan_image3_perf: - *hunyuan_image3 - - tests/dfx/perf/scripts/run_diffusion_benchmark.py + - tests/dfx/perf/scripts/run_benchmark.py - tests/dfx/perf/tests/test_hunyuan_image_tp2_cfgp2.json - tests/dfx/perf/tests/test_hunyuan_image_tp2_sp2.json - tests/dfx/perf/tests/test_hunyuan_image_tp4.json diff --git a/.buildkite/cuda/test-nightly.yml b/.buildkite/cuda/test-nightly.yml index 894d2047f09..ee3de59f8df 100644 --- a/.buildkite/cuda/test-nightly.yml +++ b/.buildkite/cuda/test-nightly.yml @@ -243,26 +243,24 @@ steps: key: nightly-diffusion-x2iat-performance-qwen-image-single timeout_in_minutes: 180 artifact_paths: - - tests/dfx/perf/results/diffusion_result_*.json - - tests/dfx/perf/results/logs/*.log + - tests/dfx/perf/results/*.json commands: - - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results + - export BENCHMARK_DIR=tests/dfx/perf/results - export CACHE_DIT_VERSION=1.5.0 - - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_1" + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_1" - label: ":full_moon: Diffusion X2I(&A&T) · Perf Test · Qwen-Image · 4-GPU" source_file_dependencies: diffusion_qwen_image_perf key: nightly-diffusion-x2iat-performance-qwen-image-4gpu timeout_in_minutes: 180 artifact_paths: - - tests/dfx/perf/results/diffusion_result_*.json - - tests/dfx/perf/results/logs/*.log + - tests/dfx/perf/results/*.json commands: - - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results + - export BENCHMARK_DIR=tests/dfx/perf/results - export CACHE_DIT_VERSION=1.5.0 # Do not pin DIFFUSION_ATTENTION_BACKEND: H100 auto-selects FLASH_ATTN; # B200 rejects explicit FLASH_ATTN without FA4 and uses CUDNN/TRTLLM. - - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_4" + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_4" - group: ":card_index_dividers: Diffusion X2V Model Test" key: nightly-diffusion-x2v-group diff --git a/.buildkite/npu/test-npu-nightly.yml b/.buildkite/npu/test-npu-nightly.yml index 4dfa2e77914..518f19f86c3 100644 --- a/.buildkite/npu/test-npu-nightly.yml +++ b/.buildkite/npu/test-npu-nightly.yml @@ -186,13 +186,12 @@ steps: HF_TOKEN: "${HF_TOKEN}" DIFFUSION_ATTENTION_BACKEND: TORCH_SDPA commands: - - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results + - export BENCHMARK_DIR=tests/dfx/perf/results - | set +e - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py -m A3 --test-config-file tests/dfx/perf/tests/test_hunyuan_image_tp2_cfgp2.json + pytest -s -v tests/dfx/perf/scripts/run_benchmark.py -m A3 --test-config-file tests/dfx/perf/tests/test_hunyuan_image_tp2_cfgp2.json EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/diffusion_result_*.json" - buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" + buildkite-agent artifact upload "tests/dfx/perf/results/*.json" exit $$EXIT - label: ":full_moon: Diffusion Perf · HunyuanImage3 TP=2 SP=2" source_file_dependencies: diffusion_hunyuan_image3_perf @@ -203,13 +202,12 @@ steps: HF_TOKEN: "${HF_TOKEN}" DIFFUSION_ATTENTION_BACKEND: TORCH_SDPA commands: - - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results + - export BENCHMARK_DIR=tests/dfx/perf/results - | set +e - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py -m A3 --test-config-file tests/dfx/perf/tests/test_hunyuan_image_tp2_sp2.json + pytest -s -v tests/dfx/perf/scripts/run_benchmark.py -m A3 --test-config-file tests/dfx/perf/tests/test_hunyuan_image_tp2_sp2.json EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/diffusion_result_*.json" - buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" + buildkite-agent artifact upload "tests/dfx/perf/results/*.json" exit $$EXIT - label: ":full_moon: Diffusion Perf · HunyuanImage3 TP=4" source_file_dependencies: diffusion_hunyuan_image3_perf @@ -220,13 +218,12 @@ steps: HF_TOKEN: "${HF_TOKEN}" DIFFUSION_ATTENTION_BACKEND: TORCH_SDPA commands: - - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results + - export BENCHMARK_DIR=tests/dfx/perf/results - | set +e - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py -m A3 --test-config-file tests/dfx/perf/tests/test_hunyuan_image_tp4.json + pytest -s -v tests/dfx/perf/scripts/run_benchmark.py -m A3 --test-config-file tests/dfx/perf/tests/test_hunyuan_image_tp4.json EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/diffusion_result_*.json" - buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" + buildkite-agent artifact upload "tests/dfx/perf/results/*.json" exit $$EXIT - group: ":card_index_dividers: Diffusion X2V Model Test" key: nightly-diffusion-x2v-group diff --git a/docs/contributing/ci/test_examples/l4_performance_tests.inc.md b/docs/contributing/ci/test_examples/l4_performance_tests.inc.md index 981e6089e3c..d560c1032c9 100644 --- a/docs/contributing/ci/test_examples/l4_performance_tests.inc.md +++ b/docs/contributing/ci/test_examples/l4_performance_tests.inc.md @@ -1,4 +1,4 @@ -When you want to add L4-level ***performance test*** cases, add entries to JSON files under `tests/dfx/perf/tests/` and run them via `tests/dfx/perf/scripts/run_benchmark.py` (omni / TTS) or `run_diffusion_benchmark.py` (diffusion). +When you want to add L4-level ***performance test*** cases, add entries to JSON files under `tests/dfx/perf/tests/` and run them via `tests/dfx/perf/scripts/run_benchmark.py` (omni / TTS / most diffusion OpenAI endpoints) or `run_diffusion_benchmark.py` (remaining diffusion-client cases such as custom jsonl). #### Config file layout (in-tree examples) @@ -7,8 +7,8 @@ When you want to add L4-level ***performance test*** cases, add entries to JSON | Omni (nightly) | `run_benchmark.py` | `test_qwen3_omni_no_async_chunk.json`, `test_qwen3_omni_async_chunk.json` (`full_model` without `slow` in `mark`) | | Omni (weekly) | `run_benchmark.py` | `test_qwen3_omni_async_chunk.json` (CUDA only), `test_qwen3_omni_vllm_text.json`, `test_qwen3_omni_multi_replicas.json` (`slow` in `mark`; **Perf Test** in `test-weekly.yml`) | | TTS | `run_benchmark.py` | `test_tts.json`, `test_voxcpm2.json`, `test_higgs_audio_v3.json` | -| Diffusion (`/v1/chat/completions`) | `run_diffusion_benchmark.py` | `test_qwen_image_vllm_omni.json`, `test_bagel_vllm_omni.json`, … | -| Diffusion (`/v1/images/generations`, `/v1/images/edits`, `/v1/videos`) | `run_benchmark.py` | `test_wan22_i2v_vllm_omni.json`, `test_cosmos3_vllm_omni.json`, `test_lingbot_video_vllm_omni.json`, … | +| Diffusion (chat / images / videos via omni bench) | `run_benchmark.py` | `test_qwen_image_vllm_omni.json`, `test_bagel_vllm_omni.json`, `test_wan22_i2v_vllm_omni.json`, `test_cosmos3_vllm_omni.json`, … | +| Diffusion (custom jsonl / remaining diffusion client) | `run_diffusion_benchmark.py` | `test_hunyuan_image3_it2i.json` | #### How runners pick cases @@ -70,7 +70,7 @@ Pass **`--test-config-file`** to load one JSON file, or omit it for the bulk sca | server_type | Diffusion | Only for diffusion-script JSON; omit on omni-bench generation cases | | benchmark_endpoint | Optional | Legacy diffusion custom-jsonl alias; prefer `benchmark_params[].endpoint` | -Omit `mark` only for configs not meant to be filtered by `-m`. Cases that call `/v1/images/edits`, `/v1/images/generations`, or `/v1/videos` use the same `benchmark_params` schema as Omni/TTS (`dataset_name`, `endpoint`, `extra_body`) and are executed by `run_benchmark.py` — do not set `server_type` or `task` on those cases. Remaining diffusion cases (usually `/v1/chat/completions`, or custom jsonl) stay on `run_diffusion_benchmark.py` and may keep `server_type`. +Omit `mark` only for configs not meant to be filtered by `-m`. Diffusion image/video cases that use OpenAI-compatible endpoints (`/v1/chat/completions`, `/v1/images/generations`, `/v1/images/edits`, `/v1/videos`) share the Omni/TTS `benchmark_params` schema (`dataset_name`, `endpoint`, `extra_body`) and run via `run_benchmark.py` — do not set `server_type` or `task`. Remaining diffusion-script cases (custom jsonl, or features not yet in omni bench such as `random-request-config`) stay on `run_diffusion_benchmark.py` and may keep `server_type`. #### `mark` field @@ -140,22 +140,28 @@ Result files use the **runtime** hardware label from `get_runtime_resource_label Examples: -- Omni/TTS and OpenAI generation endpoints (`/v1/images/*`, `/v1/videos`): `result_{test_name}_{optional_hw}_{dataset}_....json` under `BENCHMARK_DIR` +- Omni/TTS and diffusion OpenAI endpoints (`/v1/chat/completions`, `/v1/images/*`, `/v1/videos`): `result_{test_name}_{optional_hw}_{dataset}_....json` under `BENCHMARK_DIR` - Remaining diffusion (`run_diffusion_benchmark.py`): one aggregate `diffusion_result_{config_stem}_{optional_hw}_{timestamp}.json` per source JSON file #### Local commands ```bash # Bulk load + filter by JSON mark -pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py -m "full_model and H100 and diffusion" +pytest -s -v tests/dfx/perf/scripts/run_benchmark.py -m "full_model and H100 and diffusion" pytest -s -v tests/dfx/perf/scripts/run_benchmark.py -m "full_model and omni and H100" +pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py -m "full_model and H100 and diffusion" # Single file (same selectors as the CI Perf steps) -pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py \ +pytest -s -v tests/dfx/perf/scripts/run_benchmark.py \ --test-config-file tests/dfx/perf/tests/test_bagel_vllm_omni.json +pytest -s -v tests/dfx/perf/scripts/run_benchmark.py \ + --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json \ + -m "H100 and B200 and cards_1" pytest -s -v tests/dfx/perf/scripts/run_benchmark.py \ --test-config-file tests/dfx/perf/tests/test_cosmos3_vllm_omni.json \ -m "H100 and B200 and cards_2" +pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py \ + --test-config-file tests/dfx/perf/tests/test_hunyuan_image3_it2i.json pytest -s -v tests/dfx/perf/scripts/run_benchmark.py \ --test-config-file tests/dfx/perf/tests/test_qwen3_omni_async_chunk.json \ -m "H100 and full_model and not slow" @@ -214,7 +220,7 @@ You can add any benchmark running parameters you need here. For all optional par 2. For boolean variables in the running parameters, modify them to forms such as ignore_eos: true/false and fill them into the JSON file. 3. Optionally add a `baseline` object (see **Baseline thresholds** below). If you omit `baseline` or leave it empty, the performance test still runs but does not assert metric thresholds from this field. 4. Set `"name"` on each `benchmark_params` entry for stable pytest ids and readable result keys. -5. Image/video generation cases (`/v1/images/generations`, `/v1/images/edits`, `/v1/videos`) use this same schema: set `endpoint` to the API path, put width/height/steps/frames in `extra_body`, use `dataset_name: random` for text-only inputs or `random-mm` when the request needs a synthetic image/video. Do not set `server_type` or `task`, and do not use `random-request-config` or kebab-case diffusion client fields. +5. Image/video generation cases (`/v1/chat/completions`, `/v1/images/generations`, `/v1/images/edits`, `/v1/videos`) use this same schema: set `endpoint` to the API path (and `backend: openai-chat-omni` for chat), put width/height/steps/frames (and optional `negative_prompt`) in `extra_body`, use `dataset_name: random` for text-only inputs or `random-mm` when the request needs a synthetic image/video, and set `tokenizer` (e.g. `gpt2`) when the model has no HF tokenizer. Do not set `server_type` or `task`, and do not use `random-request-config` or kebab-case diffusion client fields. 6. The qps and concurrency modes are recommended to be mutually exclusive. For detailed explanations, see the table below: | Parameter | Type | Required | Example/Values | Description | diff --git a/docs/contributing/ci/test_execution_guide.md b/docs/contributing/ci/test_execution_guide.md index 774ebe41cc2..80c58a22377 100644 --- a/docs/contributing/ci/test_execution_guide.md +++ b/docs/contributing/ci/test_execution_guide.md @@ -160,11 +160,13 @@ Failed jobs: 1/2 ```bash pytest -s -v -m "full_model and L4 and not cards_1" --run-level=full_model ``` - Note: ``run_benchmark.py`` and ``run_diffusion_benchmark.py`` accept an optional ``--test-config-file``. If omitted, each loads every ``*.json`` under ``tests/dfx/perf/tests/`` (omni/tts vs diffusion split by ``is_diffusion_perf_config``) and pytest ``-m`` filters by each case's JSON ``mark``: + Note: ``run_benchmark.py`` and ``run_diffusion_benchmark.py`` accept an optional ``--test-config-file``. If omitted, each loads every ``*.json`` under ``tests/dfx/perf/tests/`` (omni/tts/generation vs remaining diffusion-client split by ``is_diffusion_perf_config``) and pytest ``-m`` filters by each case's JSON ``mark``: ```bash pytest -sv tests/dfx/perf/scripts/run_benchmark.py -m "full_model and tts and H100" + pytest -sv tests/dfx/perf/scripts/run_benchmark.py -m "full_model and diffusion and H100" pytest -sv tests/dfx/perf/scripts/run_diffusion_benchmark.py -m "full_model and diffusion and H100" pytest -sv tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_tts.json + pytest -sv tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json pytest -sv tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_cosmos3_vllm_omni.json ``` Nightly **Perf Test** jobs in [``test-nightly.yml``](https://github.com/vllm-project/vllm-omni/blob/main/.buildkite/cuda/test-nightly.yml) use ``--test-config-file`` only (no ``-m``). Weekly **Perf Test** in [``test-weekly.yml``](https://github.com/vllm-project/vllm-omni/blob/main/.buildkite/cuda/test-weekly.yml) runs ``test_qwen3_omni_vllm_text.json`` and ``test_qwen3_omni_multi_replicas.json`` (JSON ``mark`` includes ``slow``). E2e L4 function tests use ``full_model`` + ``--run-level full_model``. Example: diff --git a/docs/contributing/ci/test_writing_guide.md b/docs/contributing/ci/test_writing_guide.md index 54a6629bf78..304c4f63b77 100644 --- a/docs/contributing/ci/test_writing_guide.md +++ b/docs/contributing/ci/test_writing_guide.md @@ -160,7 +160,8 @@ When `mark` is present, it must be an **array** with exactly one ``hardware_mark ``` - Local bulk load: `pytest -sv tests/dfx/perf/scripts/run_benchmark.py -m "full_model and H100"` (omni/TTS and `/v1/images/*` + `/v1/videos` diffusion) -- Diffusion chat-completions remaining cases: `pytest -sv tests/dfx/perf/scripts/run_diffusion_benchmark.py -m "full_model and diffusion and H100"` +- Diffusion remaining custom-jsonl cases: `pytest -sv tests/dfx/perf/scripts/run_diffusion_benchmark.py -m "full_model and diffusion and H100"` +- Diffusion image/video via omni bench: `pytest -sv tests/dfx/perf/scripts/run_benchmark.py -m "full_model and diffusion and H100"` - Nightly CI perf steps: `--test-config-file tests/dfx/perf/tests/test__vllm_omni.json` (file selects cases; no `-m`) - Result filenames use **runtime** GPU detection (`get_runtime_resource_label`); `H100` is omitted on the default CI pool diff --git a/tests/dfx/perf/scripts/run_benchmark.py b/tests/dfx/perf/scripts/run_benchmark.py index 97239988dfd..ec76163cef0 100644 --- a/tests/dfx/perf/scripts/run_benchmark.py +++ b/tests/dfx/perf/scripts/run_benchmark.py @@ -61,7 +61,7 @@ def _get_config_file_from_argv() -> str | None: if skipped: print( f"--test-config-file: loaded {len(BENCHMARK_CONFIGS)} omni/tts/generation case(s); " - f"skipped {skipped} remaining diffusion case(s) (chat completions / custom jsonl)" + f"skipped {skipped} remaining diffusion case(s) (custom jsonl / diffusion-only client)" ) DEPLOY_CONFIGS_DIR = Path(__file__).parent.parent / "deploy" diff --git a/tests/dfx/perf/scripts/run_diffusion_benchmark.py b/tests/dfx/perf/scripts/run_diffusion_benchmark.py index 0523b9b3a1a..9ae6a1db57d 100644 --- a/tests/dfx/perf/scripts/run_diffusion_benchmark.py +++ b/tests/dfx/perf/scripts/run_diffusion_benchmark.py @@ -2,19 +2,25 @@ # SPDX-FileCopyrightText: Copyright contributors to the vLLM-Omni project """ -Performance benchmark CI runner for diffusion models. +Performance benchmark CI runner for remaining diffusion-client cases. + +Most image/video OpenAI endpoints now use ``run_benchmark.py`` +(``vllm bench serve --omni``). This runner keeps cases that still need the +diffusion client schema (``benchmark_params[].dataset``), for example custom +jsonl such as ``test_hunyuan_image3_it2i.json``. This runner separates two concepts: 1. ``server_type``: how the serving process is started. Currently only ``vllm-omni`` is supported here. 2. ``benchmark_endpoint``: which serving API the benchmark client calls. - Examples: ``/v1/chat/completions`` and ``/v1/videos``. + Examples: ``/v1/chat/completions`` and ``/v1/images/edits``. A config JSON file may be passed via --test-config-file. If omitted, every ``*.json`` under -``tests/dfx/perf/tests/`` is loaded and pytest ``-m`` filters by each case's ``mark``: +``tests/dfx/perf/tests/`` is loaded and only ``is_diffusion_perf_config`` cases are kept; +pytest ``-m`` filters by each case's ``mark``: pytest run_diffusion_benchmark.py -m "diffusion" - pytest run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json + pytest run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_hunyuan_image3_it2i.json Optional JSON field ``mark`` is applied as pytest marks on that case via ``pytest.param`` (e.g. ``"mark": [{"hardware_marks": {"res": {"cuda": "H100"}, "num_cards": 1}}, "full_model", "diffusion"]``). diff --git a/tests/dfx/perf/tests/test_bagel_vllm_omni.json b/tests/dfx/perf/tests/test_bagel_vllm_omni.json index 1189ddd9db7..0bb71819c69 100644 --- a/tests/dfx/perf/tests/test_bagel_vllm_omni.json +++ b/tests/dfx/perf/tests/test_bagel_vllm_omni.json @@ -14,7 +14,6 @@ "diffusion" ], "description": "Single-stage BAGEL (TP=1, CFG=1), t2i Task", - "server_type": "vllm-omni", "server_params": { "model": "ByteDance-Seed/BAGEL-7B-MoT", "serve_args": { @@ -23,7 +22,7 @@ "no-async-chunk": true, "tensor-parallel-size": 1, "cfg-parallel-size": 1, - "gpu-memory-utilization": 0.90, + "gpu-memory-utilization": 0.9, "max-model-len": 8192, "max-num-batched-tokens": 9216 } @@ -31,22 +30,28 @@ "benchmark_params": [ { "name": "512x512_t2i_steps20", - "dataset": "random", - "task": "t2i", - "width": 512, - "height": 512, - "num-inference-steps": 20, - "num-prompts": 20, - "max-concurrency": 1, - "enable-negative-prompt": true, + "num_prompts": 20, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.495, - "latency_mean": 2.0238, - "peak_memory_mb_max": 29392.0, - "peak_memory_mb_mean": 29392.0 + "request_throughput": 0.495, + "mean_e2el_ms": 2023.8, + "mean_peak_memory_mb": 29392.0 } - } + }, + "dataset_name": "random", + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 512, + "height": 512, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" } ] }, @@ -65,7 +70,6 @@ "diffusion" ], "description": "Single-stage BAGEL (TP=1, CFG=1), i2i Task", - "server_type": "vllm-omni", "server_params": { "model": "ByteDance-Seed/BAGEL-7B-MoT", "serve_args": { @@ -74,7 +78,7 @@ "no-async-chunk": true, "tensor-parallel-size": 1, "cfg-parallel-size": 1, - "gpu-memory-utilization": 0.90, + "gpu-memory-utilization": 0.9, "max-model-len": 8192, "max-num-batched-tokens": 9216 } @@ -82,23 +86,36 @@ "benchmark_params": [ { "name": "512x512_i2i_steps20", - "dataset": "random", - "task": "i2i", - "width": 512, - "height": 512, - "num-inference-steps": 20, - "num-prompts": 20, - "max-concurrency": 1, - "enable-negative-prompt": true, - "num-input-images": 1, + "num_prompts": 20, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.3499, - "latency_mean": 2.8701, - "peak_memory_mb_max": 30466.0, - "peak_memory_mb_mean": 30465.0 + "request_throughput": 0.3499, + "mean_e2el_ms": 2870.1, + "mean_peak_memory_mb": 30465.0 } - } + }, + "dataset_name": "random-mm", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(512, 512, 1)": 1.0 + }, + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 512, + "height": 512, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" } ] }, @@ -117,7 +134,6 @@ "diffusion" ], "description": "Two-stage BAGEL (bagel.yaml), TP=1, CFG=1, t2i Task", - "server_type": "vllm-omni", "server_params": { "model": "ByteDance-Seed/BAGEL-7B-MoT", "serve_args": { @@ -134,22 +150,28 @@ "benchmark_params": [ { "name": "512x512_t2i_steps20", - "dataset": "random", - "task": "t2i", - "width": 512, - "height": 512, - "num-inference-steps": 20, - "num-prompts": 20, - "max-concurrency": 1, - "enable-negative-prompt": true, + "num_prompts": 20, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.5048, - "latency_mean": 1.987, - "peak_memory_mb_max": 29380.0, - "peak_memory_mb_mean": 29380.0 + "request_throughput": 0.5048, + "mean_e2el_ms": 1987.0, + "mean_peak_memory_mb": 29380.0 } - } + }, + "dataset_name": "random", + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 512, + "height": 512, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" } ] }, @@ -168,7 +190,6 @@ "diffusion" ], "description": "Two-stage BAGEL (bagel.yaml), TP=1, CFG=1, i2i Task", - "server_type": "vllm-omni", "server_params": { "model": "ByteDance-Seed/BAGEL-7B-MoT", "serve_args": { @@ -185,23 +206,36 @@ "benchmark_params": [ { "name": "512x512_i2i_steps20", - "dataset": "random", - "task": "i2i", - "width": 512, - "height": 512, - "num-inference-steps": 20, - "num-prompts": 20, - "max-concurrency": 1, - "enable-negative-prompt": true, - "num-input-images": 1, + "num_prompts": 20, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.1533, - "latency_mean": 6.5243, - "peak_memory_mb_max": 30574.0, - "peak_memory_mb_mean": 30574.0 + "request_throughput": 0.1533, + "mean_e2el_ms": 6524.3, + "mean_peak_memory_mb": 30574.0 } - } + }, + "dataset_name": "random-mm", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(512, 512, 1)": 1.0 + }, + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 512, + "height": 512, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" } ] } diff --git a/tests/dfx/perf/tests/test_boogu_image_edit_vllm_omni.json b/tests/dfx/perf/tests/test_boogu_image_edit_vllm_omni.json index 70e0728a0a7..810bb709e5c 100644 --- a/tests/dfx/perf/tests/test_boogu_image_edit_vllm_omni.json +++ b/tests/dfx/perf/tests/test_boogu_image_edit_vllm_omni.json @@ -1,216 +1,279 @@ [ - { - "test_name": "test_boogu_image_edit_single_device", - "description": "Single-device baseline (no parallelism), concurrency sweep 1/2/4 at 512x512, 28 steps. This row measures text-only edit guidance (2 predictions/step); dedicated rows below use extra-body to measure double guidance. Baselines recorded on 1x A40 46GB (measured c1/c2/c4: qps 0.0337/0.0336/0.0336, latency_mean 29.64/56.50/101.48 s, peak mem 36754 MB) with ~10% margin.", - "server_type": "vllm-omni", - "server_params": { - "model": "Boogu/Boogu-Image-0.1-Edit", - "serve_args": { - "enable-diffusion-pipeline-profiler": true - } - }, - "benchmark_params": [ - { - "name": "512x512_steps28_i2i", - "dataset": "random", - "task": "i2i", - "width": 512, - "height": 512, - "num-inference-steps": 28, - "num-prompts": 10, - "max-concurrency": [1, 2, 4], - "warmup-requests": 4, - "warmup-concurrency": 4, - "warmup-num-inference-steps": 28, - "enable-negative-prompt": true, - "baseline": { - "H100": { - "throughput_qps": [0.0303, 0.0302, 0.0302], - "latency_mean": [32.6, 62.2, 111.7], - "peak_memory_mb_max": 40500, - "peak_memory_mb_mean": 40500 - } - } - } - ] + { + "test_name": "test_boogu_image_edit_single_device", + "description": "Single-device baseline (no parallelism), concurrency sweep 1/2/4 at 512x512, 28 steps. This row measures text-only edit guidance (2 predictions/step); dedicated rows below use extra-body to measure double guidance. Baselines recorded on 1x A40 46GB (measured c1/c2/c4: qps 0.0337/0.0336/0.0336, latency_mean 29.64/56.50/101.48 s, peak mem 36754 MB) with ~10% margin.", + "server_params": { + "model": "Boogu/Boogu-Image-0.1-Edit", + "serve_args": { + "enable-diffusion-pipeline-profiler": true + } }, - { - "test_name": "test_boogu_image_edit_double_guidance_single_device", - "mark": [ - { - "hardware_marks": { - "res": { - "cuda": "H100" - }, - "num_cards": 1 - } - }, - "full_model", - "diffusion" + "benchmark_params": [ + { + "name": "512x512_steps28_i2i", + "num_prompts": 10, + "max_concurrency": [ + 1, + 2, + 4 ], - "description": "Single-device sequential three-branch double-guidance control row for direct A/B comparison with cfg_parallel_size=2 and 3 on identical H100 hardware.", - "server_type": "vllm-omni", - "server_params": { - "model": "Boogu/Boogu-Image-0.1-Edit", - "serve_args": { - "enable-diffusion-pipeline-profiler": true - } - }, - "benchmark_params": [ - { - "name": "512x512_steps28_i2i_double_cfg1", - "dataset": "random", - "task": "i2i", - "width": 512, - "height": 512, - "num-inference-steps": 28, - "num-prompts": 10, - "max-concurrency": 1, - "warmup-requests": 4, - "warmup-concurrency": 1, - "warmup-num-inference-steps": 28, - "enable-negative-prompt": true, - "extra-body": { - "guidance_scale": 5.0, - "guidance_scale_2": 2.0, - "seed": 42 - } - } - ] + "baseline": { + "H100": { + "request_throughput": [ + 0.0303, + 0.0302, + 0.0302 + ], + "mean_e2el_ms": [ + 32600.0, + 62200.0, + 111700.0 + ], + "mean_peak_memory_mb": 40500 + } + }, + "num_warmups": 4, + "dataset_name": "random-mm", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(512, 512, 1)": 1.0 + }, + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 512, + "height": 512, + "num_inference_steps": 28, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" + } + ] + }, + { + "test_name": "test_boogu_image_edit_double_guidance_single_device", + "mark": [ + { + "hardware_marks": { + "res": { + "cuda": "H100" + }, + "num_cards": 1 + } + }, + "full_model", + "diffusion" + ], + "description": "Single-device sequential three-branch double-guidance control row for direct A/B comparison with cfg_parallel_size=2 and 3 on identical H100 hardware.", + "server_params": { + "model": "Boogu/Boogu-Image-0.1-Edit", + "serve_args": { + "enable-diffusion-pipeline-profiler": true + } }, - { - "test_name": "test_boogu_image_edit_text_cfg_parallel_2", - "mark": [ - { - "hardware_marks": { - "res": { - "cuda": "H100" - }, - "num_cards": 2 - } - }, - "full_model", - "diffusion" - ], - "description": "Two-device CFG-parallel Edit text-only guidance measurement row. Compare with the single-device row on identical H100 hardware and software.", - "server_type": "vllm-omni", - "server_params": { - "model": "Boogu/Boogu-Image-0.1-Edit", - "serve_args": { - "cfg-parallel-size": 2, - "enable-diffusion-pipeline-profiler": true - } - }, - "benchmark_params": [ - { - "name": "512x512_steps28_i2i_text_cfg2", - "dataset": "random", - "task": "i2i", - "width": 512, - "height": 512, - "num-inference-steps": 28, - "num-prompts": 10, - "max-concurrency": 1, - "warmup-requests": 4, - "warmup-concurrency": 1, - "warmup-num-inference-steps": 28, - "enable-negative-prompt": true, - "extra-body": { - "guidance_scale": 5.0, - "guidance_scale_2": 1.0, - "seed": 42 - } - } - ] + "benchmark_params": [ + { + "name": "512x512_steps28_i2i_double_cfg1", + "num_prompts": 10, + "max_concurrency": 1, + "num_warmups": 4, + "dataset_name": "random-mm", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(512, 512, 1)": 1.0 + }, + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "guidance_scale": 5.0, + "guidance_scale_2": 2.0, + "seed": 42, + "width": 512, + "height": 512, + "num_inference_steps": 28, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" + } + ] + }, + { + "test_name": "test_boogu_image_edit_text_cfg_parallel_2", + "mark": [ + { + "hardware_marks": { + "res": { + "cuda": "H100" + }, + "num_cards": 2 + } + }, + "full_model", + "diffusion" + ], + "description": "Two-device CFG-parallel Edit text-only guidance measurement row. Compare with the single-device row on identical H100 hardware and software.", + "server_params": { + "model": "Boogu/Boogu-Image-0.1-Edit", + "serve_args": { + "cfg-parallel-size": 2, + "enable-diffusion-pipeline-profiler": true + } }, - { - "test_name": "test_boogu_image_edit_double_cfg_parallel_2", - "mark": [ - { - "hardware_marks": { - "res": { - "cuda": "H100" - }, - "num_cards": 2 - } - }, - "full_model", - "diffusion" - ], - "description": "Two-device round-robin execution of the three Edit double-guidance branches. Populate A/B latency, throughput, and per-GPU memory after GPU validation.", - "server_type": "vllm-omni", - "server_params": { - "model": "Boogu/Boogu-Image-0.1-Edit", - "serve_args": { - "cfg-parallel-size": 2, - "enable-diffusion-pipeline-profiler": true - } - }, - "benchmark_params": [ - { - "name": "512x512_steps28_i2i_double_cfg2", - "dataset": "random", - "task": "i2i", - "width": 512, - "height": 512, - "num-inference-steps": 28, - "num-prompts": 10, - "max-concurrency": 1, - "warmup-requests": 4, - "warmup-concurrency": 1, - "warmup-num-inference-steps": 28, - "enable-negative-prompt": true, - "extra-body": { - "guidance_scale": 5.0, - "guidance_scale_2": 2.0, - "seed": 42 - } - } - ] + "benchmark_params": [ + { + "name": "512x512_steps28_i2i_text_cfg2", + "num_prompts": 10, + "max_concurrency": 1, + "num_warmups": 4, + "dataset_name": "random-mm", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(512, 512, 1)": 1.0 + }, + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "guidance_scale": 5.0, + "guidance_scale_2": 1.0, + "seed": 42, + "width": 512, + "height": 512, + "num_inference_steps": 28, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" + } + ] + }, + { + "test_name": "test_boogu_image_edit_double_cfg_parallel_2", + "mark": [ + { + "hardware_marks": { + "res": { + "cuda": "H100" + }, + "num_cards": 2 + } + }, + "full_model", + "diffusion" + ], + "description": "Two-device round-robin execution of the three Edit double-guidance branches. Populate A/B latency, throughput, and per-GPU memory after GPU validation.", + "server_params": { + "model": "Boogu/Boogu-Image-0.1-Edit", + "serve_args": { + "cfg-parallel-size": 2, + "enable-diffusion-pipeline-profiler": true + } }, - { - "test_name": "test_boogu_image_edit_double_cfg_parallel_3", - "mark": [ - { - "hardware_marks": { - "res": { - "cuda": "H100" - }, - "num_cards": 3 - } - }, - "full_model", - "diffusion" - ], - "description": "Full three-device execution of the three Edit double-guidance branches. Compare directly with sequential and cfg_parallel_size=2 on identical H100 hardware.", - "server_type": "vllm-omni", - "server_params": { - "model": "Boogu/Boogu-Image-0.1-Edit", - "serve_args": { - "cfg-parallel-size": 3, - "enable-diffusion-pipeline-profiler": true - } - }, - "benchmark_params": [ - { - "name": "512x512_steps28_i2i_double_cfg3", - "dataset": "random", - "task": "i2i", - "width": 512, - "height": 512, - "num-inference-steps": 28, - "num-prompts": 10, - "max-concurrency": 1, - "warmup-requests": 4, - "warmup-concurrency": 1, - "warmup-num-inference-steps": 28, - "enable-negative-prompt": true, - "extra-body": { - "guidance_scale": 5.0, - "guidance_scale_2": 2.0, - "seed": 42 - } - } - ] - } + "benchmark_params": [ + { + "name": "512x512_steps28_i2i_double_cfg2", + "num_prompts": 10, + "max_concurrency": 1, + "num_warmups": 4, + "dataset_name": "random-mm", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(512, 512, 1)": 1.0 + }, + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "guidance_scale": 5.0, + "guidance_scale_2": 2.0, + "seed": 42, + "width": 512, + "height": 512, + "num_inference_steps": 28, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" + } + ] + }, + { + "test_name": "test_boogu_image_edit_double_cfg_parallel_3", + "mark": [ + { + "hardware_marks": { + "res": { + "cuda": "H100" + }, + "num_cards": 3 + } + }, + "full_model", + "diffusion" + ], + "description": "Full three-device execution of the three Edit double-guidance branches. Compare directly with sequential and cfg_parallel_size=2 on identical H100 hardware.", + "server_params": { + "model": "Boogu/Boogu-Image-0.1-Edit", + "serve_args": { + "cfg-parallel-size": 3, + "enable-diffusion-pipeline-profiler": true + } + }, + "benchmark_params": [ + { + "name": "512x512_steps28_i2i_double_cfg3", + "num_prompts": 10, + "max_concurrency": 1, + "num_warmups": 4, + "dataset_name": "random-mm", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(512, 512, 1)": 1.0 + }, + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "guidance_scale": 5.0, + "guidance_scale_2": 2.0, + "seed": 42, + "width": 512, + "height": 512, + "num_inference_steps": 28, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" + } + ] + } ] diff --git a/tests/dfx/perf/tests/test_boogu_image_vllm_omni.json b/tests/dfx/perf/tests/test_boogu_image_vllm_omni.json index b0f35bd4aed..ef462e19a4f 100644 --- a/tests/dfx/perf/tests/test_boogu_image_vllm_omni.json +++ b/tests/dfx/perf/tests/test_boogu_image_vllm_omni.json @@ -1,81 +1,98 @@ [ - { - "test_name": "test_boogu_image_single_device", - "description": "Single-device baseline (no parallelism), concurrency sweep 1/2/4 at 512x512, 28 steps. Baselines recorded on 1x A40 46GB (measured c1/c2/c4: qps 0.0611/0.0609/0.0611, latency_mean 16.37/31.22/55.77 s, peak mem 36754 MB) with ~10% margin.", - "server_type": "vllm-omni", - "server_params": { - "model": "Boogu/Boogu-Image-0.1-Base", - "serve_args": { - "enable-diffusion-pipeline-profiler": true - } - }, - "benchmark_params": [ - { - "name": "512x512_steps28", - "dataset": "random", - "task": "t2i", - "width": 512, - "height": 512, - "num-inference-steps": 28, - "num-prompts": 10, - "max-concurrency": [1, 2, 4], - "warmup-requests": 4, - "warmup-concurrency": 4, - "warmup-num-inference-steps": 28, - "enable-negative-prompt": true, - "baseline": { - "H100": { - "throughput_qps": [0.0549, 0.0547, 0.0549], - "latency_mean": [18.0, 34.3, 61.4], - "peak_memory_mb_max": 40500, - "peak_memory_mb_mean": 40500 - } - } - } - ] + { + "test_name": "test_boogu_image_single_device", + "description": "Single-device baseline (no parallelism), concurrency sweep 1/2/4 at 512x512, 28 steps. Baselines recorded on 1x A40 46GB (measured c1/c2/c4: qps 0.0611/0.0609/0.0611, latency_mean 16.37/31.22/55.77 s, peak mem 36754 MB) with ~10% margin.", + "server_params": { + "model": "Boogu/Boogu-Image-0.1-Base", + "serve_args": { + "enable-diffusion-pipeline-profiler": true + } }, - { - "test_name": "test_boogu_image_cfg_parallel_2", - "mark": [ - { - "hardware_marks": { - "res": { - "cuda": "H100" - }, - "num_cards": 2 - } - }, - "full_model", - "diffusion" + "benchmark_params": [ + { + "name": "512x512_steps28", + "num_prompts": 10, + "max_concurrency": [ + 1, + 2, + 4 ], - "description": "Two-device CFG-parallel Base T2I measurement row. Run beside test_boogu_image_single_device on the same H100 revision and populate latency, throughput, speedup, and per-GPU peak-memory baselines from the resulting report.", - "server_type": "vllm-omni", - "server_params": { - "model": "Boogu/Boogu-Image-0.1-Base", - "serve_args": { - "cfg-parallel-size": 2, - "enable-diffusion-pipeline-profiler": true - } + "baseline": { + "H100": { + "request_throughput": [ + 0.0549, + 0.0547, + 0.0549 + ], + "mean_e2el_ms": [ + 18000.0, + 34300.0, + 61400.0 + ], + "mean_peak_memory_mb": 40500 + } + }, + "num_warmups": 4, + "dataset_name": "random", + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 512, + "height": 512, + "num_inference_steps": 28, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" + } + ] + }, + { + "test_name": "test_boogu_image_cfg_parallel_2", + "mark": [ + { + "hardware_marks": { + "res": { + "cuda": "H100" + }, + "num_cards": 2 + } + }, + "full_model", + "diffusion" + ], + "description": "Two-device CFG-parallel Base T2I measurement row. Run beside test_boogu_image_single_device on the same H100 revision and populate latency, throughput, speedup, and per-GPU peak-memory baselines from the resulting report.", + "server_params": { + "model": "Boogu/Boogu-Image-0.1-Base", + "serve_args": { + "cfg-parallel-size": 2, + "enable-diffusion-pipeline-profiler": true + } + }, + "benchmark_params": [ + { + "name": "512x512_steps28_cfg2", + "num_prompts": 10, + "max_concurrency": 1, + "num_warmups": 4, + "dataset_name": "random", + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "guidance_scale": 4.0, + "seed": 42, + "width": 512, + "height": 512, + "num_inference_steps": 28, + "negative_prompt": "Negative prompt for benchmarking diffusion models" }, - "benchmark_params": [ - { - "name": "512x512_steps28_cfg2", - "dataset": "random", - "task": "t2i", - "width": 512, - "height": 512, - "num-inference-steps": 28, - "num-prompts": 10, - "max-concurrency": 1, - "warmup-requests": 4, - "warmup-concurrency": 1, - "warmup-num-inference-steps": 28, - "enable-negative-prompt": true, - "extra-body": { - "guidance_scale": 4.0, - "seed": 42 - } - } - ] - } + "backend": "openai-chat-omni" + } + ] + } ] diff --git a/tests/dfx/perf/tests/test_hunyuan_image_tp2_cfgp2.json b/tests/dfx/perf/tests/test_hunyuan_image_tp2_cfgp2.json index a86a6a3d575..8a59452cb53 100644 --- a/tests/dfx/perf/tests/test_hunyuan_image_tp2_cfgp2.json +++ b/tests/dfx/perf/tests/test_hunyuan_image_tp2_cfgp2.json @@ -15,7 +15,6 @@ "local_model" ], "description": "TP=2 CfgP=2 baseline", - "server_type": "vllm-omni", "server_params": { "model": "tencent/HunyuanImage-3.0-Instruct", "serve_args": { @@ -30,21 +29,27 @@ "benchmark_params": [ { "name": "1024x1024_steps8", - "dataset": "random", - "task": "t2i", - "width": 1024, - "height": 1024, - "num-inference-steps": 8, - "num-prompts": 10, - "max-concurrency": 1, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H200": { - "throughput_qps": 0.21, - "latency_p99": 4.7469, - "peak_memory_mb_max": 101100, - "peak_memory_mb_mean": 101100 + "request_throughput": 0.21, + "mean_peak_memory_mb": 101100, + "mean_e2el_ms": 4746.9 } - } + }, + "dataset_name": "random", + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 1024, + "height": 1024, + "num_inference_steps": 8 + }, + "backend": "openai-chat-omni" } ] } diff --git a/tests/dfx/perf/tests/test_hunyuan_image_tp2_sp2.json b/tests/dfx/perf/tests/test_hunyuan_image_tp2_sp2.json index e49cd0543ca..8d833f77ec0 100644 --- a/tests/dfx/perf/tests/test_hunyuan_image_tp2_sp2.json +++ b/tests/dfx/perf/tests/test_hunyuan_image_tp2_sp2.json @@ -15,7 +15,6 @@ "local_model" ], "description": "TP=2 SP=2 baseline", - "server_type": "vllm-omni", "server_params": { "model": "tencent/HunyuanImage-3.0-Instruct", "serve_args": { @@ -30,21 +29,27 @@ "benchmark_params": [ { "name": "1024x1024_steps8", - "dataset": "random", - "task": "t2i", - "width": 1024, - "height": 1024, - "num-inference-steps": 8, - "num-prompts": 10, - "max-concurrency": 1, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H200": { - "throughput_qps": 0.20, - "latency_p99": 5.1025, - "peak_memory_mb_max": 97402, - "peak_memory_mb_mean": 97402 + "request_throughput": 0.2, + "mean_peak_memory_mb": 97402, + "mean_e2el_ms": 5102.5 } - } + }, + "dataset_name": "random", + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 1024, + "height": 1024, + "num_inference_steps": 8 + }, + "backend": "openai-chat-omni" } ] } diff --git a/tests/dfx/perf/tests/test_hunyuan_image_tp4.json b/tests/dfx/perf/tests/test_hunyuan_image_tp4.json index 54631471a38..ae143092a96 100644 --- a/tests/dfx/perf/tests/test_hunyuan_image_tp4.json +++ b/tests/dfx/perf/tests/test_hunyuan_image_tp4.json @@ -15,7 +15,6 @@ "local_model" ], "description": "TP=4 baseline", - "server_type": "vllm-omni", "server_params": { "model": "tencent/HunyuanImage-3.0-Instruct", "serve_args": { @@ -29,22 +28,27 @@ "benchmark_params": [ { "name": "1024x1024_steps8", - "dataset": "random", - "task": "t2i", - "width": 1024, - "height": 1024, - "num-inference-steps": 8, - "num-prompts": 10, - "max-concurrency": 1, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H200": { - "throughput_qps": 0.213, - "latency_mean": 4.6953, - "latency_p99": 4.8601, - "peak_memory_mb_max": 56912.0, - "peak_memory_mb_mean": 56912.0 + "request_throughput": 0.213, + "mean_e2el_ms": 4695.3, + "mean_peak_memory_mb": 56912.0 } - } + }, + "dataset_name": "random", + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 1024, + "height": 1024, + "num_inference_steps": 8 + }, + "backend": "openai-chat-omni" } ] } diff --git a/tests/dfx/perf/tests/test_qwen_image_edit_2511_vllm_omni.json b/tests/dfx/perf/tests/test_qwen_image_edit_2511_vllm_omni.json index 34d275ac971..f9d4c921956 100644 --- a/tests/dfx/perf/tests/test_qwen_image_edit_2511_vllm_omni.json +++ b/tests/dfx/perf/tests/test_qwen_image_edit_2511_vllm_omni.json @@ -14,7 +14,6 @@ "diffusion" ], "description": "Single-device baseline (two input images)", - "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image-Edit-2511", "serve_args": { @@ -24,43 +23,69 @@ "benchmark_params": [ { "name": "512x512_steps20_i2i_2img", - "dataset": "random", - "task": "i2i", - "width": 512, - "height": 512, - "num-inference-steps": 20, - "num-prompts": 10, - "max-concurrency": 1, - "num-input-images": 2, - "enable-negative-prompt": true, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.0684, - "latency_mean": 14.6249, - "peak_memory_mb_max": 57632.0, - "peak_memory_mb_mean": 57632.0 + "request_throughput": 0.0684, + "mean_e2el_ms": 14624.9, + "mean_peak_memory_mb": 57632.0 } - } + }, + "dataset_name": "random-mm", + "random_mm_base_items_per_request": 2, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 2 + }, + "random_mm_bucket_config": { + "(512, 512, 1)": 1.0 + }, + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 512, + "height": 512, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" }, { "name": "1536x1536_steps35_i2i_2img", - "dataset": "random", - "task": "i2i", - "width": 1536, - "height": 1536, - "num-inference-steps": 35, - "num-prompts": 10, - "max-concurrency": 1, - "num-input-images": 2, - "enable-negative-prompt": true, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.0177, - "latency_mean": 56.5285, - "peak_memory_mb_max": 67840.0, - "peak_memory_mb_mean": 67840.0 + "request_throughput": 0.0177, + "mean_e2el_ms": 56528.5, + "mean_peak_memory_mb": 67840.0 } - } + }, + "dataset_name": "random-mm", + "random_mm_base_items_per_request": 2, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 2 + }, + "random_mm_bucket_config": { + "(512, 512, 1)": 1.0 + }, + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 1536, + "height": 1536, + "num_inference_steps": 35, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" } ] }, @@ -80,7 +105,6 @@ "diffusion" ], "description": "Ulysses SP=2 + CFG=2 + VAE patch parallel=4", - "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image-Edit-2511", "serve_args": { @@ -94,23 +118,36 @@ "benchmark_params": [ { "name": "1536x1536_steps35_i2i_2img", - "dataset": "random", - "task": "i2i", - "width": 1536, - "height": 1536, - "num-inference-steps": 35, - "num-prompts": 10, - "max-concurrency": 1, - "num-input-images": 2, - "enable-negative-prompt": true, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.0515, - "latency_mean": 19.4345, - "peak_memory_mb_max": 56348.0, - "peak_memory_mb_mean": 56348.0 + "request_throughput": 0.0515, + "mean_e2el_ms": 19434.5, + "mean_peak_memory_mb": 56348.0 } - } + }, + "dataset_name": "random-mm", + "random_mm_base_items_per_request": 2, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 2 + }, + "random_mm_bucket_config": { + "(512, 512, 1)": 1.0 + }, + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 1536, + "height": 1536, + "num_inference_steps": 35, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" } ] }, @@ -129,7 +166,6 @@ "diffusion" ], "description": "Ulysses SP=2 + CFG=2 + CacheDiT", - "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image-Edit-2511", "serve_args": { @@ -153,43 +189,69 @@ "benchmark_params": [ { "name": "512x512_steps20_i2i_2img", - "dataset": "random", - "task": "i2i", - "width": 512, - "height": 512, - "num-inference-steps": 20, - "num-prompts": 10, - "max-concurrency": 1, - "num-input-images": 2, - "enable-negative-prompt": true, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.2479, - "latency_mean": 4.0389, - "peak_memory_mb_max": 57750.0, - "peak_memory_mb_mean": 57750.0 + "request_throughput": 0.2479, + "mean_e2el_ms": 4038.9, + "mean_peak_memory_mb": 57750.0 } - } + }, + "dataset_name": "random-mm", + "random_mm_base_items_per_request": 2, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 2 + }, + "random_mm_bucket_config": { + "(512, 512, 1)": 1.0 + }, + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 512, + "height": 512, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" }, { "name": "1536x1536_steps35_i2i_2img", - "dataset": "random", - "task": "i2i", - "width": 1536, - "height": 1536, - "num-inference-steps": 35, - "num-prompts": 10, - "max-concurrency": 1, - "num-input-images": 2, - "enable-negative-prompt": true, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.0948, - "latency_mean": 10.5462, - "peak_memory_mb_max": 67764.0, - "peak_memory_mb_mean": 67764.0 + "request_throughput": 0.0948, + "mean_e2el_ms": 10546.2, + "mean_peak_memory_mb": 67764.0 } - } + }, + "dataset_name": "random-mm", + "random_mm_base_items_per_request": 2, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 2 + }, + "random_mm_bucket_config": { + "(512, 512, 1)": 1.0 + }, + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 1536, + "height": 1536, + "num_inference_steps": 35, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" } ] } diff --git a/tests/dfx/perf/tests/test_qwen_image_layered_vllm_omni.json b/tests/dfx/perf/tests/test_qwen_image_layered_vllm_omni.json index bc0f5c9239f..19f02d403c0 100644 --- a/tests/dfx/perf/tests/test_qwen_image_layered_vllm_omni.json +++ b/tests/dfx/perf/tests/test_qwen_image_layered_vllm_omni.json @@ -14,7 +14,6 @@ "diffusion" ], "description": "Single-device baseline", - "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image-Layered", "serve_args": { @@ -24,41 +23,69 @@ "benchmark_params": [ { "name": "640x640_steps20_i2i", - "dataset": "random", - "task": "i2i", - "width": 640, - "height": 640, - "num-inference-steps": 20, - "num-prompts": 10, - "max-concurrency": 1, - "enable-negative-prompt": true, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.0703, - "latency_mean": 14.208, - "peak_memory_mb_max": 63225.1429, - "peak_memory_mb_mean": 63225.1429 + "request_throughput": 0.0703, + "mean_e2el_ms": 14208.0, + "mean_peak_memory_mb": 63225.1429 } - } + }, + "dataset_name": "random-mm", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(512, 512, 1)": 1.0 + }, + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 640, + "height": 640, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" }, { "name": "1024x1024_steps35_i2i", - "dataset": "random", - "task": "i2i", - "width": 1024, - "height": 1024, - "num-inference-steps": 35, - "num-prompts": 10, - "max-concurrency": 1, - "enable-negative-prompt": true, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.0417, - "latency_mean": 23.914, - "peak_memory_mb_max": 62912.0, - "peak_memory_mb_mean": 62912.0 + "request_throughput": 0.0417, + "mean_e2el_ms": 23914.0, + "mean_peak_memory_mb": 62912.0 } - } + }, + "dataset_name": "random-mm", + "random_mm_base_items_per_request": 1, + "random_mm_num_mm_items_range_ratio": 0, + "random_mm_limit_mm_per_prompt": { + "image": 1 + }, + "random_mm_bucket_config": { + "(512, 512, 1)": 1.0 + }, + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 1024, + "height": 1024, + "num_inference_steps": 35, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" } ] } diff --git a/tests/dfx/perf/tests/test_qwen_image_vllm_omni.json b/tests/dfx/perf/tests/test_qwen_image_vllm_omni.json index 61cb334eb7b..237cc3c5b6f 100644 --- a/tests/dfx/perf/tests/test_qwen_image_vllm_omni.json +++ b/tests/dfx/perf/tests/test_qwen_image_vllm_omni.json @@ -5,7 +5,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 1 } @@ -14,7 +17,6 @@ "diffusion" ], "description": "Single-device baseline (no parallelism)", - "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image", "serve_args": { @@ -24,39 +26,53 @@ "benchmark_params": [ { "name": "512x512_steps20", - "dataset": "random", - "task": "t2i", - "width": 512, - "height": 512, - "num-inference-steps": 20, - "num-prompts": 10, - "max-concurrency": 1, - "enable-negative-prompt": true, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.4153, - "latency_mean": 2.4065, - "peak_memory_mb_mean": 55062.0 + "request_throughput": 0.4153, + "mean_e2el_ms": 2406.5, + "mean_peak_memory_mb": 55062.0 } - } + }, + "dataset_name": "random", + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 512, + "height": 512, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" }, { "name": "1536x1536_steps35", - "dataset": "random", - "task": "t2i", - "width": 1536, - "height": 1536, - "num-inference-steps": 35, - "num-prompts": 10, - "max-concurrency": 1, - "enable-negative-prompt": true, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.0423, - "latency_mean": 23.5991, - "peak_memory_mb_mean": 64546.0 + "request_throughput": 0.0423, + "mean_e2el_ms": 23599.1, + "mean_peak_memory_mb": 64546.0 } - } + }, + "dataset_name": "random", + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 1536, + "height": 1536, + "num_inference_steps": 35, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" } ] }, @@ -66,7 +82,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 1 } @@ -75,7 +94,6 @@ "diffusion" ], "description": "Single-device step execution: sequential 512/1536 plus 512 concurrency sweep. One server (max-num-seqs 8).", - "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image", "serve_args": { @@ -87,60 +105,104 @@ "benchmark_params": [ { "name": "512x512_steps20", - "dataset": "random", - "task": "t2i", - "width": 512, - "height": 512, - "num-inference-steps": 20, - "num-prompts": 10, - "max-concurrency": 1, - "enable-negative-prompt": true, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.4175, - "latency_mean": 2.3953, - "peak_memory_mb_mean": 54952.0 + "request_throughput": 0.4175, + "mean_e2el_ms": 2395.3, + "mean_peak_memory_mb": 54952.0 } - } + }, + "dataset_name": "random", + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 512, + "height": 512, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" }, { "name": "1536x1536_steps35", - "dataset": "random", - "task": "t2i", - "width": 1536, - "height": 1536, - "num-inference-steps": 35, - "num-prompts": 10, - "max-concurrency": 1, - "enable-negative-prompt": true, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.0423, - "latency_mean": 23.5768, - "peak_memory_mb_mean": 64500.0 + "request_throughput": 0.0423, + "mean_e2el_ms": 23576.8, + "mean_peak_memory_mb": 64500.0 } - } + }, + "dataset_name": "random", + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 1536, + "height": 1536, + "num_inference_steps": 35, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" }, { "name": "512x512_steps20_high_concurrency", - "dataset": "random", - "task": "t2i", - "width": 512, - "height": 512, - "num-inference-steps": 20, - "num-prompts": [20, 40, 80, 160], - "max-concurrency": [1, 2, 4, 8], - "warmup-requests": 8, - "warmup-concurrency": 8, - "warmup-num-inference-steps": 20, - "enable-negative-prompt": true, + "num_prompts": [ + 20, + 40, + 80, + 160 + ], + "max_concurrency": [ + 1, + 2, + 4, + 8 + ], "baseline": { "H100": { - "throughput_qps": [0.3937, 0.6606, 0.6188, 0.7752], - "latency_mean": [2.5466, 3.0235, 6.3948, 10.1944], - "peak_memory_mb_mean": [55149.0, 55150.0, 55153.1458, 55156.325] + "request_throughput": [ + 0.3937, + 0.6606, + 0.6188, + 0.7752 + ], + "mean_e2el_ms": [ + 2546.6, + 3023.5, + 6394.8, + 10194.4 + ], + "mean_peak_memory_mb": [ + 55149.0, + 55150.0, + 55153.1458, + 55156.325 + ] } - } + }, + "num_warmups": 8, + "dataset_name": "random", + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 512, + "height": 512, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" } ] }, @@ -150,7 +212,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 4 } @@ -159,7 +224,6 @@ "diffusion" ], "description": "Ulysses SP=2 + CFG-parallel=2 + VAE Patch Parallel=4", - "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image", "serve_args": { @@ -173,21 +237,28 @@ "benchmark_params": [ { "name": "1536x1536_steps35", - "dataset": "random", - "task": "t2i", - "width": 1536, - "height": 1536, - "num-inference-steps": 35, - "num-prompts": 10, - "max-concurrency": 1, - "enable-negative-prompt": true, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.1214, - "latency_mean": 8.2211, - "peak_memory_mb_mean": 54474.0 + "request_throughput": 0.1214, + "mean_e2el_ms": 8221.1, + "mean_peak_memory_mb": 54474.0 } - } + }, + "dataset_name": "random", + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 1536, + "height": 1536, + "num_inference_steps": 35, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" } ] }, @@ -197,7 +268,10 @@ { "hardware_marks": { "res": { - "cuda": ["H100", "B200"] + "cuda": [ + "H100", + "B200" + ] }, "num_cards": 4 } @@ -206,7 +280,6 @@ "diffusion" ], "description": "Ulysses SP=2 + CFG-parallel=2 + CacheDiT acceleration", - "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image", "serve_args": { @@ -230,39 +303,53 @@ "benchmark_params": [ { "name": "512x512_steps20", - "dataset": "random", - "task": "t2i", - "width": 512, - "height": 512, - "num-inference-steps": 20, - "num-prompts": 10, - "max-concurrency": 1, - "enable-negative-prompt": true, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.559, - "latency_mean": 1.794, - "peak_memory_mb_mean": 55174.0 + "request_throughput": 0.559, + "mean_e2el_ms": 1794.0, + "mean_peak_memory_mb": 55174.0 } - } + }, + "dataset_name": "random", + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 512, + "height": 512, + "num_inference_steps": 20, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" }, { "name": "1536x1536_steps35", - "dataset": "random", - "task": "t2i", - "width": 1536, - "height": 1536, - "num-inference-steps": 35, - "num-prompts": 10, - "max-concurrency": 1, - "enable-negative-prompt": true, + "num_prompts": 10, + "max_concurrency": 1, "baseline": { "H100": { - "throughput_qps": 0.1918, - "latency_mean": 5.2022, - "peak_memory_mb_mean": 64750.0 + "request_throughput": 0.1918, + "mean_e2el_ms": 5202.2, + "mean_peak_memory_mb": 64750.0 } - } + }, + "dataset_name": "random", + "endpoint": "/v1/chat/completions", + "tokenizer": "gpt2", + "random_input_len": 8, + "random_output_len": 1, + "percentile_metrics": "e2el", + "extra_body": { + "width": 1536, + "height": 1536, + "num_inference_steps": 35, + "negative_prompt": "Negative prompt for benchmarking diffusion models" + }, + "backend": "openai-chat-omni" } ] } From 5bf7701ba5b8bd3f931041893e1339705b44ab85 Mon Sep 17 00:00:00 2001 From: wangyu <410167048@qq.com> Date: Fri, 18 Sep 2026 16:57:46 +0800 Subject: [PATCH 07/15] Update performance benchmark scripts and configurations for diffusion models - Renamed `run_benchmark.py` to `run_diffusion_benchmark.py` in CI configurations and test files to clearly differentiate between omni and diffusion benchmarks. - Adjusted artifact paths and environment variable names for consistency across performance test configurations. - Enhanced test cases to include `server_type` for diffusion models and updated benchmark parameters for clarity and uniformity. - Improved documentation to reflect changes in test execution and configuration, ensuring better guidance for contributors. These updates aim to streamline the benchmarking process and enhance the maintainability of the codebase. Signed-off-by: wangyu <410167048@qq.com> --- .../common/ci_source_file_dependencies.yml | 4 +- .buildkite/cuda/test-nightly.yml | 14 +- .buildkite/npu/test-npu-nightly.yml | 21 +- .../test_examples/l4_performance_tests.inc.md | 22 +- docs/contributing/ci/test_execution_guide.md | 4 +- docs/contributing/ci/test_writing_guide.md | 3 +- tests/dfx/perf/scripts/run_benchmark.py | 2 +- .../perf/scripts/run_diffusion_benchmark.py | 14 +- .../dfx/perf/tests/test_bagel_vllm_omni.json | 154 +++--- .../test_boogu_image_edit_vllm_omni.json | 481 ++++++++---------- .../tests/test_boogu_image_vllm_omni.json | 169 +++--- .../tests/test_hunyuan_image_tp2_cfgp2.json | 31 +- .../tests/test_hunyuan_image_tp2_sp2.json | 31 +- .../perf/tests/test_hunyuan_image_tp4.json | 32 +- .../test_qwen_image_edit_2511_vllm_omni.json | 208 +++----- .../test_qwen_image_layered_vllm_omni.json | 81 +-- .../perf/tests/test_qwen_image_vllm_omni.json | 301 ++++------- vllm_omni/benchmarks/patch/patch.py | 1 + .../models/cosmos3/pipeline_cosmos3.py | 18 +- .../pipeline_hunyuan_video_1_5.py | 7 +- .../pipeline_hunyuan_video_1_5_i2v.py | 7 +- 21 files changed, 655 insertions(+), 950 deletions(-) diff --git a/.buildkite/common/ci_source_file_dependencies.yml b/.buildkite/common/ci_source_file_dependencies.yml index 8ac6945360a..6795e9678b7 100644 --- a/.buildkite/common/ci_source_file_dependencies.yml +++ b/.buildkite/common/ci_source_file_dependencies.yml @@ -338,7 +338,7 @@ source_file_dependencies: diffusion_qwen_image_perf: - *qwen_image - - tests/dfx/perf/scripts/run_benchmark.py + - tests/dfx/perf/scripts/run_diffusion_benchmark.py - tests/dfx/perf/tests/test_qwen_image_vllm_omni.json diffusion_text_to_image_doc: @@ -370,7 +370,7 @@ source_file_dependencies: diffusion_hunyuan_image3_perf: - *hunyuan_image3 - - tests/dfx/perf/scripts/run_benchmark.py + - tests/dfx/perf/scripts/run_diffusion_benchmark.py - tests/dfx/perf/tests/test_hunyuan_image_tp2_cfgp2.json - tests/dfx/perf/tests/test_hunyuan_image_tp2_sp2.json - tests/dfx/perf/tests/test_hunyuan_image_tp4.json diff --git a/.buildkite/cuda/test-nightly.yml b/.buildkite/cuda/test-nightly.yml index ee3de59f8df..894d2047f09 100644 --- a/.buildkite/cuda/test-nightly.yml +++ b/.buildkite/cuda/test-nightly.yml @@ -243,24 +243,26 @@ steps: key: nightly-diffusion-x2iat-performance-qwen-image-single timeout_in_minutes: 180 artifact_paths: - - tests/dfx/perf/results/*.json + - tests/dfx/perf/results/diffusion_result_*.json + - tests/dfx/perf/results/logs/*.log commands: - - export BENCHMARK_DIR=tests/dfx/perf/results + - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results - export CACHE_DIT_VERSION=1.5.0 - - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_1" + - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_1" - label: ":full_moon: Diffusion X2I(&A&T) · Perf Test · Qwen-Image · 4-GPU" source_file_dependencies: diffusion_qwen_image_perf key: nightly-diffusion-x2iat-performance-qwen-image-4gpu timeout_in_minutes: 180 artifact_paths: - - tests/dfx/perf/results/*.json + - tests/dfx/perf/results/diffusion_result_*.json + - tests/dfx/perf/results/logs/*.log commands: - - export BENCHMARK_DIR=tests/dfx/perf/results + - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results - export CACHE_DIT_VERSION=1.5.0 # Do not pin DIFFUSION_ATTENTION_BACKEND: H100 auto-selects FLASH_ATTN; # B200 rejects explicit FLASH_ATTN without FA4 and uses CUDNN/TRTLLM. - - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_4" + - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_4" - group: ":card_index_dividers: Diffusion X2V Model Test" key: nightly-diffusion-x2v-group diff --git a/.buildkite/npu/test-npu-nightly.yml b/.buildkite/npu/test-npu-nightly.yml index 518f19f86c3..4dfa2e77914 100644 --- a/.buildkite/npu/test-npu-nightly.yml +++ b/.buildkite/npu/test-npu-nightly.yml @@ -186,12 +186,13 @@ steps: HF_TOKEN: "${HF_TOKEN}" DIFFUSION_ATTENTION_BACKEND: TORCH_SDPA commands: - - export BENCHMARK_DIR=tests/dfx/perf/results + - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results - | set +e - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py -m A3 --test-config-file tests/dfx/perf/tests/test_hunyuan_image_tp2_cfgp2.json + pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py -m A3 --test-config-file tests/dfx/perf/tests/test_hunyuan_image_tp2_cfgp2.json EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/*.json" + buildkite-agent artifact upload "tests/dfx/perf/results/diffusion_result_*.json" + buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" exit $$EXIT - label: ":full_moon: Diffusion Perf · HunyuanImage3 TP=2 SP=2" source_file_dependencies: diffusion_hunyuan_image3_perf @@ -202,12 +203,13 @@ steps: HF_TOKEN: "${HF_TOKEN}" DIFFUSION_ATTENTION_BACKEND: TORCH_SDPA commands: - - export BENCHMARK_DIR=tests/dfx/perf/results + - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results - | set +e - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py -m A3 --test-config-file tests/dfx/perf/tests/test_hunyuan_image_tp2_sp2.json + pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py -m A3 --test-config-file tests/dfx/perf/tests/test_hunyuan_image_tp2_sp2.json EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/*.json" + buildkite-agent artifact upload "tests/dfx/perf/results/diffusion_result_*.json" + buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" exit $$EXIT - label: ":full_moon: Diffusion Perf · HunyuanImage3 TP=4" source_file_dependencies: diffusion_hunyuan_image3_perf @@ -218,12 +220,13 @@ steps: HF_TOKEN: "${HF_TOKEN}" DIFFUSION_ATTENTION_BACKEND: TORCH_SDPA commands: - - export BENCHMARK_DIR=tests/dfx/perf/results + - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results - | set +e - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py -m A3 --test-config-file tests/dfx/perf/tests/test_hunyuan_image_tp4.json + pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py -m A3 --test-config-file tests/dfx/perf/tests/test_hunyuan_image_tp4.json EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/*.json" + buildkite-agent artifact upload "tests/dfx/perf/results/diffusion_result_*.json" + buildkite-agent artifact upload "tests/dfx/perf/results/logs/*.log" exit $$EXIT - group: ":card_index_dividers: Diffusion X2V Model Test" key: nightly-diffusion-x2v-group diff --git a/docs/contributing/ci/test_examples/l4_performance_tests.inc.md b/docs/contributing/ci/test_examples/l4_performance_tests.inc.md index d560c1032c9..981e6089e3c 100644 --- a/docs/contributing/ci/test_examples/l4_performance_tests.inc.md +++ b/docs/contributing/ci/test_examples/l4_performance_tests.inc.md @@ -1,4 +1,4 @@ -When you want to add L4-level ***performance test*** cases, add entries to JSON files under `tests/dfx/perf/tests/` and run them via `tests/dfx/perf/scripts/run_benchmark.py` (omni / TTS / most diffusion OpenAI endpoints) or `run_diffusion_benchmark.py` (remaining diffusion-client cases such as custom jsonl). +When you want to add L4-level ***performance test*** cases, add entries to JSON files under `tests/dfx/perf/tests/` and run them via `tests/dfx/perf/scripts/run_benchmark.py` (omni / TTS) or `run_diffusion_benchmark.py` (diffusion). #### Config file layout (in-tree examples) @@ -7,8 +7,8 @@ When you want to add L4-level ***performance test*** cases, add entries to JSON | Omni (nightly) | `run_benchmark.py` | `test_qwen3_omni_no_async_chunk.json`, `test_qwen3_omni_async_chunk.json` (`full_model` without `slow` in `mark`) | | Omni (weekly) | `run_benchmark.py` | `test_qwen3_omni_async_chunk.json` (CUDA only), `test_qwen3_omni_vllm_text.json`, `test_qwen3_omni_multi_replicas.json` (`slow` in `mark`; **Perf Test** in `test-weekly.yml`) | | TTS | `run_benchmark.py` | `test_tts.json`, `test_voxcpm2.json`, `test_higgs_audio_v3.json` | -| Diffusion (chat / images / videos via omni bench) | `run_benchmark.py` | `test_qwen_image_vllm_omni.json`, `test_bagel_vllm_omni.json`, `test_wan22_i2v_vllm_omni.json`, `test_cosmos3_vllm_omni.json`, … | -| Diffusion (custom jsonl / remaining diffusion client) | `run_diffusion_benchmark.py` | `test_hunyuan_image3_it2i.json` | +| Diffusion (`/v1/chat/completions`) | `run_diffusion_benchmark.py` | `test_qwen_image_vllm_omni.json`, `test_bagel_vllm_omni.json`, … | +| Diffusion (`/v1/images/generations`, `/v1/images/edits`, `/v1/videos`) | `run_benchmark.py` | `test_wan22_i2v_vllm_omni.json`, `test_cosmos3_vllm_omni.json`, `test_lingbot_video_vllm_omni.json`, … | #### How runners pick cases @@ -70,7 +70,7 @@ Pass **`--test-config-file`** to load one JSON file, or omit it for the bulk sca | server_type | Diffusion | Only for diffusion-script JSON; omit on omni-bench generation cases | | benchmark_endpoint | Optional | Legacy diffusion custom-jsonl alias; prefer `benchmark_params[].endpoint` | -Omit `mark` only for configs not meant to be filtered by `-m`. Diffusion image/video cases that use OpenAI-compatible endpoints (`/v1/chat/completions`, `/v1/images/generations`, `/v1/images/edits`, `/v1/videos`) share the Omni/TTS `benchmark_params` schema (`dataset_name`, `endpoint`, `extra_body`) and run via `run_benchmark.py` — do not set `server_type` or `task`. Remaining diffusion-script cases (custom jsonl, or features not yet in omni bench such as `random-request-config`) stay on `run_diffusion_benchmark.py` and may keep `server_type`. +Omit `mark` only for configs not meant to be filtered by `-m`. Cases that call `/v1/images/edits`, `/v1/images/generations`, or `/v1/videos` use the same `benchmark_params` schema as Omni/TTS (`dataset_name`, `endpoint`, `extra_body`) and are executed by `run_benchmark.py` — do not set `server_type` or `task` on those cases. Remaining diffusion cases (usually `/v1/chat/completions`, or custom jsonl) stay on `run_diffusion_benchmark.py` and may keep `server_type`. #### `mark` field @@ -140,28 +140,22 @@ Result files use the **runtime** hardware label from `get_runtime_resource_label Examples: -- Omni/TTS and diffusion OpenAI endpoints (`/v1/chat/completions`, `/v1/images/*`, `/v1/videos`): `result_{test_name}_{optional_hw}_{dataset}_....json` under `BENCHMARK_DIR` +- Omni/TTS and OpenAI generation endpoints (`/v1/images/*`, `/v1/videos`): `result_{test_name}_{optional_hw}_{dataset}_....json` under `BENCHMARK_DIR` - Remaining diffusion (`run_diffusion_benchmark.py`): one aggregate `diffusion_result_{config_stem}_{optional_hw}_{timestamp}.json` per source JSON file #### Local commands ```bash # Bulk load + filter by JSON mark -pytest -s -v tests/dfx/perf/scripts/run_benchmark.py -m "full_model and H100 and diffusion" -pytest -s -v tests/dfx/perf/scripts/run_benchmark.py -m "full_model and omni and H100" pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py -m "full_model and H100 and diffusion" +pytest -s -v tests/dfx/perf/scripts/run_benchmark.py -m "full_model and omni and H100" # Single file (same selectors as the CI Perf steps) -pytest -s -v tests/dfx/perf/scripts/run_benchmark.py \ +pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py \ --test-config-file tests/dfx/perf/tests/test_bagel_vllm_omni.json -pytest -s -v tests/dfx/perf/scripts/run_benchmark.py \ - --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json \ - -m "H100 and B200 and cards_1" pytest -s -v tests/dfx/perf/scripts/run_benchmark.py \ --test-config-file tests/dfx/perf/tests/test_cosmos3_vllm_omni.json \ -m "H100 and B200 and cards_2" -pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py \ - --test-config-file tests/dfx/perf/tests/test_hunyuan_image3_it2i.json pytest -s -v tests/dfx/perf/scripts/run_benchmark.py \ --test-config-file tests/dfx/perf/tests/test_qwen3_omni_async_chunk.json \ -m "H100 and full_model and not slow" @@ -220,7 +214,7 @@ You can add any benchmark running parameters you need here. For all optional par 2. For boolean variables in the running parameters, modify them to forms such as ignore_eos: true/false and fill them into the JSON file. 3. Optionally add a `baseline` object (see **Baseline thresholds** below). If you omit `baseline` or leave it empty, the performance test still runs but does not assert metric thresholds from this field. 4. Set `"name"` on each `benchmark_params` entry for stable pytest ids and readable result keys. -5. Image/video generation cases (`/v1/chat/completions`, `/v1/images/generations`, `/v1/images/edits`, `/v1/videos`) use this same schema: set `endpoint` to the API path (and `backend: openai-chat-omni` for chat), put width/height/steps/frames (and optional `negative_prompt`) in `extra_body`, use `dataset_name: random` for text-only inputs or `random-mm` when the request needs a synthetic image/video, and set `tokenizer` (e.g. `gpt2`) when the model has no HF tokenizer. Do not set `server_type` or `task`, and do not use `random-request-config` or kebab-case diffusion client fields. +5. Image/video generation cases (`/v1/images/generations`, `/v1/images/edits`, `/v1/videos`) use this same schema: set `endpoint` to the API path, put width/height/steps/frames in `extra_body`, use `dataset_name: random` for text-only inputs or `random-mm` when the request needs a synthetic image/video. Do not set `server_type` or `task`, and do not use `random-request-config` or kebab-case diffusion client fields. 6. The qps and concurrency modes are recommended to be mutually exclusive. For detailed explanations, see the table below: | Parameter | Type | Required | Example/Values | Description | diff --git a/docs/contributing/ci/test_execution_guide.md b/docs/contributing/ci/test_execution_guide.md index 80c58a22377..774ebe41cc2 100644 --- a/docs/contributing/ci/test_execution_guide.md +++ b/docs/contributing/ci/test_execution_guide.md @@ -160,13 +160,11 @@ Failed jobs: 1/2 ```bash pytest -s -v -m "full_model and L4 and not cards_1" --run-level=full_model ``` - Note: ``run_benchmark.py`` and ``run_diffusion_benchmark.py`` accept an optional ``--test-config-file``. If omitted, each loads every ``*.json`` under ``tests/dfx/perf/tests/`` (omni/tts/generation vs remaining diffusion-client split by ``is_diffusion_perf_config``) and pytest ``-m`` filters by each case's JSON ``mark``: + Note: ``run_benchmark.py`` and ``run_diffusion_benchmark.py`` accept an optional ``--test-config-file``. If omitted, each loads every ``*.json`` under ``tests/dfx/perf/tests/`` (omni/tts vs diffusion split by ``is_diffusion_perf_config``) and pytest ``-m`` filters by each case's JSON ``mark``: ```bash pytest -sv tests/dfx/perf/scripts/run_benchmark.py -m "full_model and tts and H100" - pytest -sv tests/dfx/perf/scripts/run_benchmark.py -m "full_model and diffusion and H100" pytest -sv tests/dfx/perf/scripts/run_diffusion_benchmark.py -m "full_model and diffusion and H100" pytest -sv tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_tts.json - pytest -sv tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json pytest -sv tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_cosmos3_vllm_omni.json ``` Nightly **Perf Test** jobs in [``test-nightly.yml``](https://github.com/vllm-project/vllm-omni/blob/main/.buildkite/cuda/test-nightly.yml) use ``--test-config-file`` only (no ``-m``). Weekly **Perf Test** in [``test-weekly.yml``](https://github.com/vllm-project/vllm-omni/blob/main/.buildkite/cuda/test-weekly.yml) runs ``test_qwen3_omni_vllm_text.json`` and ``test_qwen3_omni_multi_replicas.json`` (JSON ``mark`` includes ``slow``). E2e L4 function tests use ``full_model`` + ``--run-level full_model``. Example: diff --git a/docs/contributing/ci/test_writing_guide.md b/docs/contributing/ci/test_writing_guide.md index 304c4f63b77..54a6629bf78 100644 --- a/docs/contributing/ci/test_writing_guide.md +++ b/docs/contributing/ci/test_writing_guide.md @@ -160,8 +160,7 @@ When `mark` is present, it must be an **array** with exactly one ``hardware_mark ``` - Local bulk load: `pytest -sv tests/dfx/perf/scripts/run_benchmark.py -m "full_model and H100"` (omni/TTS and `/v1/images/*` + `/v1/videos` diffusion) -- Diffusion remaining custom-jsonl cases: `pytest -sv tests/dfx/perf/scripts/run_diffusion_benchmark.py -m "full_model and diffusion and H100"` -- Diffusion image/video via omni bench: `pytest -sv tests/dfx/perf/scripts/run_benchmark.py -m "full_model and diffusion and H100"` +- Diffusion chat-completions remaining cases: `pytest -sv tests/dfx/perf/scripts/run_diffusion_benchmark.py -m "full_model and diffusion and H100"` - Nightly CI perf steps: `--test-config-file tests/dfx/perf/tests/test__vllm_omni.json` (file selects cases; no `-m`) - Result filenames use **runtime** GPU detection (`get_runtime_resource_label`); `H100` is omitted on the default CI pool diff --git a/tests/dfx/perf/scripts/run_benchmark.py b/tests/dfx/perf/scripts/run_benchmark.py index ec76163cef0..97239988dfd 100644 --- a/tests/dfx/perf/scripts/run_benchmark.py +++ b/tests/dfx/perf/scripts/run_benchmark.py @@ -61,7 +61,7 @@ def _get_config_file_from_argv() -> str | None: if skipped: print( f"--test-config-file: loaded {len(BENCHMARK_CONFIGS)} omni/tts/generation case(s); " - f"skipped {skipped} remaining diffusion case(s) (custom jsonl / diffusion-only client)" + f"skipped {skipped} remaining diffusion case(s) (chat completions / custom jsonl)" ) DEPLOY_CONFIGS_DIR = Path(__file__).parent.parent / "deploy" diff --git a/tests/dfx/perf/scripts/run_diffusion_benchmark.py b/tests/dfx/perf/scripts/run_diffusion_benchmark.py index 9ae6a1db57d..0523b9b3a1a 100644 --- a/tests/dfx/perf/scripts/run_diffusion_benchmark.py +++ b/tests/dfx/perf/scripts/run_diffusion_benchmark.py @@ -2,25 +2,19 @@ # SPDX-FileCopyrightText: Copyright contributors to the vLLM-Omni project """ -Performance benchmark CI runner for remaining diffusion-client cases. - -Most image/video OpenAI endpoints now use ``run_benchmark.py`` -(``vllm bench serve --omni``). This runner keeps cases that still need the -diffusion client schema (``benchmark_params[].dataset``), for example custom -jsonl such as ``test_hunyuan_image3_it2i.json``. +Performance benchmark CI runner for diffusion models. This runner separates two concepts: 1. ``server_type``: how the serving process is started. Currently only ``vllm-omni`` is supported here. 2. ``benchmark_endpoint``: which serving API the benchmark client calls. - Examples: ``/v1/chat/completions`` and ``/v1/images/edits``. + Examples: ``/v1/chat/completions`` and ``/v1/videos``. A config JSON file may be passed via --test-config-file. If omitted, every ``*.json`` under -``tests/dfx/perf/tests/`` is loaded and only ``is_diffusion_perf_config`` cases are kept; -pytest ``-m`` filters by each case's ``mark``: +``tests/dfx/perf/tests/`` is loaded and pytest ``-m`` filters by each case's ``mark``: pytest run_diffusion_benchmark.py -m "diffusion" - pytest run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_hunyuan_image3_it2i.json + pytest run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json Optional JSON field ``mark`` is applied as pytest marks on that case via ``pytest.param`` (e.g. ``"mark": [{"hardware_marks": {"res": {"cuda": "H100"}, "num_cards": 1}}, "full_model", "diffusion"]``). diff --git a/tests/dfx/perf/tests/test_bagel_vllm_omni.json b/tests/dfx/perf/tests/test_bagel_vllm_omni.json index 0bb71819c69..1189ddd9db7 100644 --- a/tests/dfx/perf/tests/test_bagel_vllm_omni.json +++ b/tests/dfx/perf/tests/test_bagel_vllm_omni.json @@ -14,6 +14,7 @@ "diffusion" ], "description": "Single-stage BAGEL (TP=1, CFG=1), t2i Task", + "server_type": "vllm-omni", "server_params": { "model": "ByteDance-Seed/BAGEL-7B-MoT", "serve_args": { @@ -22,7 +23,7 @@ "no-async-chunk": true, "tensor-parallel-size": 1, "cfg-parallel-size": 1, - "gpu-memory-utilization": 0.9, + "gpu-memory-utilization": 0.90, "max-model-len": 8192, "max-num-batched-tokens": 9216 } @@ -30,28 +31,22 @@ "benchmark_params": [ { "name": "512x512_t2i_steps20", - "num_prompts": 20, - "max_concurrency": 1, + "dataset": "random", + "task": "t2i", + "width": 512, + "height": 512, + "num-inference-steps": 20, + "num-prompts": 20, + "max-concurrency": 1, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.495, - "mean_e2el_ms": 2023.8, - "mean_peak_memory_mb": 29392.0 + "throughput_qps": 0.495, + "latency_mean": 2.0238, + "peak_memory_mb_max": 29392.0, + "peak_memory_mb_mean": 29392.0 } - }, - "dataset_name": "random", - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 512, - "height": 512, - "num_inference_steps": 20, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } } ] }, @@ -70,6 +65,7 @@ "diffusion" ], "description": "Single-stage BAGEL (TP=1, CFG=1), i2i Task", + "server_type": "vllm-omni", "server_params": { "model": "ByteDance-Seed/BAGEL-7B-MoT", "serve_args": { @@ -78,7 +74,7 @@ "no-async-chunk": true, "tensor-parallel-size": 1, "cfg-parallel-size": 1, - "gpu-memory-utilization": 0.9, + "gpu-memory-utilization": 0.90, "max-model-len": 8192, "max-num-batched-tokens": 9216 } @@ -86,36 +82,23 @@ "benchmark_params": [ { "name": "512x512_i2i_steps20", - "num_prompts": 20, - "max_concurrency": 1, + "dataset": "random", + "task": "i2i", + "width": 512, + "height": 512, + "num-inference-steps": 20, + "num-prompts": 20, + "max-concurrency": 1, + "enable-negative-prompt": true, + "num-input-images": 1, "baseline": { "H100": { - "request_throughput": 0.3499, - "mean_e2el_ms": 2870.1, - "mean_peak_memory_mb": 30465.0 + "throughput_qps": 0.3499, + "latency_mean": 2.8701, + "peak_memory_mb_max": 30466.0, + "peak_memory_mb_mean": 30465.0 } - }, - "dataset_name": "random-mm", - "random_mm_base_items_per_request": 1, - "random_mm_num_mm_items_range_ratio": 0, - "random_mm_limit_mm_per_prompt": { - "image": 1 - }, - "random_mm_bucket_config": { - "(512, 512, 1)": 1.0 - }, - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 512, - "height": 512, - "num_inference_steps": 20, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } } ] }, @@ -134,6 +117,7 @@ "diffusion" ], "description": "Two-stage BAGEL (bagel.yaml), TP=1, CFG=1, t2i Task", + "server_type": "vllm-omni", "server_params": { "model": "ByteDance-Seed/BAGEL-7B-MoT", "serve_args": { @@ -150,28 +134,22 @@ "benchmark_params": [ { "name": "512x512_t2i_steps20", - "num_prompts": 20, - "max_concurrency": 1, + "dataset": "random", + "task": "t2i", + "width": 512, + "height": 512, + "num-inference-steps": 20, + "num-prompts": 20, + "max-concurrency": 1, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.5048, - "mean_e2el_ms": 1987.0, - "mean_peak_memory_mb": 29380.0 + "throughput_qps": 0.5048, + "latency_mean": 1.987, + "peak_memory_mb_max": 29380.0, + "peak_memory_mb_mean": 29380.0 } - }, - "dataset_name": "random", - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 512, - "height": 512, - "num_inference_steps": 20, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } } ] }, @@ -190,6 +168,7 @@ "diffusion" ], "description": "Two-stage BAGEL (bagel.yaml), TP=1, CFG=1, i2i Task", + "server_type": "vllm-omni", "server_params": { "model": "ByteDance-Seed/BAGEL-7B-MoT", "serve_args": { @@ -206,36 +185,23 @@ "benchmark_params": [ { "name": "512x512_i2i_steps20", - "num_prompts": 20, - "max_concurrency": 1, + "dataset": "random", + "task": "i2i", + "width": 512, + "height": 512, + "num-inference-steps": 20, + "num-prompts": 20, + "max-concurrency": 1, + "enable-negative-prompt": true, + "num-input-images": 1, "baseline": { "H100": { - "request_throughput": 0.1533, - "mean_e2el_ms": 6524.3, - "mean_peak_memory_mb": 30574.0 + "throughput_qps": 0.1533, + "latency_mean": 6.5243, + "peak_memory_mb_max": 30574.0, + "peak_memory_mb_mean": 30574.0 } - }, - "dataset_name": "random-mm", - "random_mm_base_items_per_request": 1, - "random_mm_num_mm_items_range_ratio": 0, - "random_mm_limit_mm_per_prompt": { - "image": 1 - }, - "random_mm_bucket_config": { - "(512, 512, 1)": 1.0 - }, - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 512, - "height": 512, - "num_inference_steps": 20, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } } ] } diff --git a/tests/dfx/perf/tests/test_boogu_image_edit_vllm_omni.json b/tests/dfx/perf/tests/test_boogu_image_edit_vllm_omni.json index 810bb709e5c..70e0728a0a7 100644 --- a/tests/dfx/perf/tests/test_boogu_image_edit_vllm_omni.json +++ b/tests/dfx/perf/tests/test_boogu_image_edit_vllm_omni.json @@ -1,279 +1,216 @@ [ - { - "test_name": "test_boogu_image_edit_single_device", - "description": "Single-device baseline (no parallelism), concurrency sweep 1/2/4 at 512x512, 28 steps. This row measures text-only edit guidance (2 predictions/step); dedicated rows below use extra-body to measure double guidance. Baselines recorded on 1x A40 46GB (measured c1/c2/c4: qps 0.0337/0.0336/0.0336, latency_mean 29.64/56.50/101.48 s, peak mem 36754 MB) with ~10% margin.", - "server_params": { - "model": "Boogu/Boogu-Image-0.1-Edit", - "serve_args": { - "enable-diffusion-pipeline-profiler": true - } + { + "test_name": "test_boogu_image_edit_single_device", + "description": "Single-device baseline (no parallelism), concurrency sweep 1/2/4 at 512x512, 28 steps. This row measures text-only edit guidance (2 predictions/step); dedicated rows below use extra-body to measure double guidance. Baselines recorded on 1x A40 46GB (measured c1/c2/c4: qps 0.0337/0.0336/0.0336, latency_mean 29.64/56.50/101.48 s, peak mem 36754 MB) with ~10% margin.", + "server_type": "vllm-omni", + "server_params": { + "model": "Boogu/Boogu-Image-0.1-Edit", + "serve_args": { + "enable-diffusion-pipeline-profiler": true + } + }, + "benchmark_params": [ + { + "name": "512x512_steps28_i2i", + "dataset": "random", + "task": "i2i", + "width": 512, + "height": 512, + "num-inference-steps": 28, + "num-prompts": 10, + "max-concurrency": [1, 2, 4], + "warmup-requests": 4, + "warmup-concurrency": 4, + "warmup-num-inference-steps": 28, + "enable-negative-prompt": true, + "baseline": { + "H100": { + "throughput_qps": [0.0303, 0.0302, 0.0302], + "latency_mean": [32.6, 62.2, 111.7], + "peak_memory_mb_max": 40500, + "peak_memory_mb_mean": 40500 + } + } + } + ] }, - "benchmark_params": [ - { - "name": "512x512_steps28_i2i", - "num_prompts": 10, - "max_concurrency": [ - 1, - 2, - 4 + { + "test_name": "test_boogu_image_edit_double_guidance_single_device", + "mark": [ + { + "hardware_marks": { + "res": { + "cuda": "H100" + }, + "num_cards": 1 + } + }, + "full_model", + "diffusion" ], - "baseline": { - "H100": { - "request_throughput": [ - 0.0303, - 0.0302, - 0.0302 - ], - "mean_e2el_ms": [ - 32600.0, - 62200.0, - 111700.0 - ], - "mean_peak_memory_mb": 40500 - } - }, - "num_warmups": 4, - "dataset_name": "random-mm", - "random_mm_base_items_per_request": 1, - "random_mm_num_mm_items_range_ratio": 0, - "random_mm_limit_mm_per_prompt": { - "image": 1 - }, - "random_mm_bucket_config": { - "(512, 512, 1)": 1.0 - }, - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 512, - "height": 512, - "num_inference_steps": 28, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" - } - ] - }, - { - "test_name": "test_boogu_image_edit_double_guidance_single_device", - "mark": [ - { - "hardware_marks": { - "res": { - "cuda": "H100" - }, - "num_cards": 1 - } - }, - "full_model", - "diffusion" - ], - "description": "Single-device sequential three-branch double-guidance control row for direct A/B comparison with cfg_parallel_size=2 and 3 on identical H100 hardware.", - "server_params": { - "model": "Boogu/Boogu-Image-0.1-Edit", - "serve_args": { - "enable-diffusion-pipeline-profiler": true - } - }, - "benchmark_params": [ - { - "name": "512x512_steps28_i2i_double_cfg1", - "num_prompts": 10, - "max_concurrency": 1, - "num_warmups": 4, - "dataset_name": "random-mm", - "random_mm_base_items_per_request": 1, - "random_mm_num_mm_items_range_ratio": 0, - "random_mm_limit_mm_per_prompt": { - "image": 1 - }, - "random_mm_bucket_config": { - "(512, 512, 1)": 1.0 - }, - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "guidance_scale": 5.0, - "guidance_scale_2": 2.0, - "seed": 42, - "width": 512, - "height": 512, - "num_inference_steps": 28, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" - } - ] - }, - { - "test_name": "test_boogu_image_edit_text_cfg_parallel_2", - "mark": [ - { - "hardware_marks": { - "res": { - "cuda": "H100" - }, - "num_cards": 2 - } - }, - "full_model", - "diffusion" - ], - "description": "Two-device CFG-parallel Edit text-only guidance measurement row. Compare with the single-device row on identical H100 hardware and software.", - "server_params": { - "model": "Boogu/Boogu-Image-0.1-Edit", - "serve_args": { - "cfg-parallel-size": 2, - "enable-diffusion-pipeline-profiler": true - } + "description": "Single-device sequential three-branch double-guidance control row for direct A/B comparison with cfg_parallel_size=2 and 3 on identical H100 hardware.", + "server_type": "vllm-omni", + "server_params": { + "model": "Boogu/Boogu-Image-0.1-Edit", + "serve_args": { + "enable-diffusion-pipeline-profiler": true + } + }, + "benchmark_params": [ + { + "name": "512x512_steps28_i2i_double_cfg1", + "dataset": "random", + "task": "i2i", + "width": 512, + "height": 512, + "num-inference-steps": 28, + "num-prompts": 10, + "max-concurrency": 1, + "warmup-requests": 4, + "warmup-concurrency": 1, + "warmup-num-inference-steps": 28, + "enable-negative-prompt": true, + "extra-body": { + "guidance_scale": 5.0, + "guidance_scale_2": 2.0, + "seed": 42 + } + } + ] }, - "benchmark_params": [ - { - "name": "512x512_steps28_i2i_text_cfg2", - "num_prompts": 10, - "max_concurrency": 1, - "num_warmups": 4, - "dataset_name": "random-mm", - "random_mm_base_items_per_request": 1, - "random_mm_num_mm_items_range_ratio": 0, - "random_mm_limit_mm_per_prompt": { - "image": 1 - }, - "random_mm_bucket_config": { - "(512, 512, 1)": 1.0 - }, - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "guidance_scale": 5.0, - "guidance_scale_2": 1.0, - "seed": 42, - "width": 512, - "height": 512, - "num_inference_steps": 28, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" - } - ] - }, - { - "test_name": "test_boogu_image_edit_double_cfg_parallel_2", - "mark": [ - { - "hardware_marks": { - "res": { - "cuda": "H100" - }, - "num_cards": 2 - } - }, - "full_model", - "diffusion" - ], - "description": "Two-device round-robin execution of the three Edit double-guidance branches. Populate A/B latency, throughput, and per-GPU memory after GPU validation.", - "server_params": { - "model": "Boogu/Boogu-Image-0.1-Edit", - "serve_args": { - "cfg-parallel-size": 2, - "enable-diffusion-pipeline-profiler": true - } + { + "test_name": "test_boogu_image_edit_text_cfg_parallel_2", + "mark": [ + { + "hardware_marks": { + "res": { + "cuda": "H100" + }, + "num_cards": 2 + } + }, + "full_model", + "diffusion" + ], + "description": "Two-device CFG-parallel Edit text-only guidance measurement row. Compare with the single-device row on identical H100 hardware and software.", + "server_type": "vllm-omni", + "server_params": { + "model": "Boogu/Boogu-Image-0.1-Edit", + "serve_args": { + "cfg-parallel-size": 2, + "enable-diffusion-pipeline-profiler": true + } + }, + "benchmark_params": [ + { + "name": "512x512_steps28_i2i_text_cfg2", + "dataset": "random", + "task": "i2i", + "width": 512, + "height": 512, + "num-inference-steps": 28, + "num-prompts": 10, + "max-concurrency": 1, + "warmup-requests": 4, + "warmup-concurrency": 1, + "warmup-num-inference-steps": 28, + "enable-negative-prompt": true, + "extra-body": { + "guidance_scale": 5.0, + "guidance_scale_2": 1.0, + "seed": 42 + } + } + ] }, - "benchmark_params": [ - { - "name": "512x512_steps28_i2i_double_cfg2", - "num_prompts": 10, - "max_concurrency": 1, - "num_warmups": 4, - "dataset_name": "random-mm", - "random_mm_base_items_per_request": 1, - "random_mm_num_mm_items_range_ratio": 0, - "random_mm_limit_mm_per_prompt": { - "image": 1 - }, - "random_mm_bucket_config": { - "(512, 512, 1)": 1.0 - }, - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "guidance_scale": 5.0, - "guidance_scale_2": 2.0, - "seed": 42, - "width": 512, - "height": 512, - "num_inference_steps": 28, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" - } - ] - }, - { - "test_name": "test_boogu_image_edit_double_cfg_parallel_3", - "mark": [ - { - "hardware_marks": { - "res": { - "cuda": "H100" - }, - "num_cards": 3 - } - }, - "full_model", - "diffusion" - ], - "description": "Full three-device execution of the three Edit double-guidance branches. Compare directly with sequential and cfg_parallel_size=2 on identical H100 hardware.", - "server_params": { - "model": "Boogu/Boogu-Image-0.1-Edit", - "serve_args": { - "cfg-parallel-size": 3, - "enable-diffusion-pipeline-profiler": true - } + { + "test_name": "test_boogu_image_edit_double_cfg_parallel_2", + "mark": [ + { + "hardware_marks": { + "res": { + "cuda": "H100" + }, + "num_cards": 2 + } + }, + "full_model", + "diffusion" + ], + "description": "Two-device round-robin execution of the three Edit double-guidance branches. Populate A/B latency, throughput, and per-GPU memory after GPU validation.", + "server_type": "vllm-omni", + "server_params": { + "model": "Boogu/Boogu-Image-0.1-Edit", + "serve_args": { + "cfg-parallel-size": 2, + "enable-diffusion-pipeline-profiler": true + } + }, + "benchmark_params": [ + { + "name": "512x512_steps28_i2i_double_cfg2", + "dataset": "random", + "task": "i2i", + "width": 512, + "height": 512, + "num-inference-steps": 28, + "num-prompts": 10, + "max-concurrency": 1, + "warmup-requests": 4, + "warmup-concurrency": 1, + "warmup-num-inference-steps": 28, + "enable-negative-prompt": true, + "extra-body": { + "guidance_scale": 5.0, + "guidance_scale_2": 2.0, + "seed": 42 + } + } + ] }, - "benchmark_params": [ - { - "name": "512x512_steps28_i2i_double_cfg3", - "num_prompts": 10, - "max_concurrency": 1, - "num_warmups": 4, - "dataset_name": "random-mm", - "random_mm_base_items_per_request": 1, - "random_mm_num_mm_items_range_ratio": 0, - "random_mm_limit_mm_per_prompt": { - "image": 1 - }, - "random_mm_bucket_config": { - "(512, 512, 1)": 1.0 - }, - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "guidance_scale": 5.0, - "guidance_scale_2": 2.0, - "seed": 42, - "width": 512, - "height": 512, - "num_inference_steps": 28, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" - } - ] - } + { + "test_name": "test_boogu_image_edit_double_cfg_parallel_3", + "mark": [ + { + "hardware_marks": { + "res": { + "cuda": "H100" + }, + "num_cards": 3 + } + }, + "full_model", + "diffusion" + ], + "description": "Full three-device execution of the three Edit double-guidance branches. Compare directly with sequential and cfg_parallel_size=2 on identical H100 hardware.", + "server_type": "vllm-omni", + "server_params": { + "model": "Boogu/Boogu-Image-0.1-Edit", + "serve_args": { + "cfg-parallel-size": 3, + "enable-diffusion-pipeline-profiler": true + } + }, + "benchmark_params": [ + { + "name": "512x512_steps28_i2i_double_cfg3", + "dataset": "random", + "task": "i2i", + "width": 512, + "height": 512, + "num-inference-steps": 28, + "num-prompts": 10, + "max-concurrency": 1, + "warmup-requests": 4, + "warmup-concurrency": 1, + "warmup-num-inference-steps": 28, + "enable-negative-prompt": true, + "extra-body": { + "guidance_scale": 5.0, + "guidance_scale_2": 2.0, + "seed": 42 + } + } + ] + } ] diff --git a/tests/dfx/perf/tests/test_boogu_image_vllm_omni.json b/tests/dfx/perf/tests/test_boogu_image_vllm_omni.json index ef462e19a4f..b0f35bd4aed 100644 --- a/tests/dfx/perf/tests/test_boogu_image_vllm_omni.json +++ b/tests/dfx/perf/tests/test_boogu_image_vllm_omni.json @@ -1,98 +1,81 @@ [ - { - "test_name": "test_boogu_image_single_device", - "description": "Single-device baseline (no parallelism), concurrency sweep 1/2/4 at 512x512, 28 steps. Baselines recorded on 1x A40 46GB (measured c1/c2/c4: qps 0.0611/0.0609/0.0611, latency_mean 16.37/31.22/55.77 s, peak mem 36754 MB) with ~10% margin.", - "server_params": { - "model": "Boogu/Boogu-Image-0.1-Base", - "serve_args": { - "enable-diffusion-pipeline-profiler": true - } - }, - "benchmark_params": [ - { - "name": "512x512_steps28", - "num_prompts": 10, - "max_concurrency": [ - 1, - 2, - 4 - ], - "baseline": { - "H100": { - "request_throughput": [ - 0.0549, - 0.0547, - 0.0549 - ], - "mean_e2el_ms": [ - 18000.0, - 34300.0, - 61400.0 - ], - "mean_peak_memory_mb": 40500 - } + { + "test_name": "test_boogu_image_single_device", + "description": "Single-device baseline (no parallelism), concurrency sweep 1/2/4 at 512x512, 28 steps. Baselines recorded on 1x A40 46GB (measured c1/c2/c4: qps 0.0611/0.0609/0.0611, latency_mean 16.37/31.22/55.77 s, peak mem 36754 MB) with ~10% margin.", + "server_type": "vllm-omni", + "server_params": { + "model": "Boogu/Boogu-Image-0.1-Base", + "serve_args": { + "enable-diffusion-pipeline-profiler": true + } }, - "num_warmups": 4, - "dataset_name": "random", - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 512, - "height": 512, - "num_inference_steps": 28, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" - } - ] - }, - { - "test_name": "test_boogu_image_cfg_parallel_2", - "mark": [ - { - "hardware_marks": { - "res": { - "cuda": "H100" - }, - "num_cards": 2 - } - }, - "full_model", - "diffusion" - ], - "description": "Two-device CFG-parallel Base T2I measurement row. Run beside test_boogu_image_single_device on the same H100 revision and populate latency, throughput, speedup, and per-GPU peak-memory baselines from the resulting report.", - "server_params": { - "model": "Boogu/Boogu-Image-0.1-Base", - "serve_args": { - "cfg-parallel-size": 2, - "enable-diffusion-pipeline-profiler": true - } + "benchmark_params": [ + { + "name": "512x512_steps28", + "dataset": "random", + "task": "t2i", + "width": 512, + "height": 512, + "num-inference-steps": 28, + "num-prompts": 10, + "max-concurrency": [1, 2, 4], + "warmup-requests": 4, + "warmup-concurrency": 4, + "warmup-num-inference-steps": 28, + "enable-negative-prompt": true, + "baseline": { + "H100": { + "throughput_qps": [0.0549, 0.0547, 0.0549], + "latency_mean": [18.0, 34.3, 61.4], + "peak_memory_mb_max": 40500, + "peak_memory_mb_mean": 40500 + } + } + } + ] }, - "benchmark_params": [ - { - "name": "512x512_steps28_cfg2", - "num_prompts": 10, - "max_concurrency": 1, - "num_warmups": 4, - "dataset_name": "random", - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "guidance_scale": 4.0, - "seed": 42, - "width": 512, - "height": 512, - "num_inference_steps": 28, - "negative_prompt": "Negative prompt for benchmarking diffusion models" + { + "test_name": "test_boogu_image_cfg_parallel_2", + "mark": [ + { + "hardware_marks": { + "res": { + "cuda": "H100" + }, + "num_cards": 2 + } + }, + "full_model", + "diffusion" + ], + "description": "Two-device CFG-parallel Base T2I measurement row. Run beside test_boogu_image_single_device on the same H100 revision and populate latency, throughput, speedup, and per-GPU peak-memory baselines from the resulting report.", + "server_type": "vllm-omni", + "server_params": { + "model": "Boogu/Boogu-Image-0.1-Base", + "serve_args": { + "cfg-parallel-size": 2, + "enable-diffusion-pipeline-profiler": true + } }, - "backend": "openai-chat-omni" - } - ] - } + "benchmark_params": [ + { + "name": "512x512_steps28_cfg2", + "dataset": "random", + "task": "t2i", + "width": 512, + "height": 512, + "num-inference-steps": 28, + "num-prompts": 10, + "max-concurrency": 1, + "warmup-requests": 4, + "warmup-concurrency": 1, + "warmup-num-inference-steps": 28, + "enable-negative-prompt": true, + "extra-body": { + "guidance_scale": 4.0, + "seed": 42 + } + } + ] + } ] diff --git a/tests/dfx/perf/tests/test_hunyuan_image_tp2_cfgp2.json b/tests/dfx/perf/tests/test_hunyuan_image_tp2_cfgp2.json index 8a59452cb53..a86a6a3d575 100644 --- a/tests/dfx/perf/tests/test_hunyuan_image_tp2_cfgp2.json +++ b/tests/dfx/perf/tests/test_hunyuan_image_tp2_cfgp2.json @@ -15,6 +15,7 @@ "local_model" ], "description": "TP=2 CfgP=2 baseline", + "server_type": "vllm-omni", "server_params": { "model": "tencent/HunyuanImage-3.0-Instruct", "serve_args": { @@ -29,27 +30,21 @@ "benchmark_params": [ { "name": "1024x1024_steps8", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "t2i", + "width": 1024, + "height": 1024, + "num-inference-steps": 8, + "num-prompts": 10, + "max-concurrency": 1, "baseline": { "H200": { - "request_throughput": 0.21, - "mean_peak_memory_mb": 101100, - "mean_e2el_ms": 4746.9 + "throughput_qps": 0.21, + "latency_p99": 4.7469, + "peak_memory_mb_max": 101100, + "peak_memory_mb_mean": 101100 } - }, - "dataset_name": "random", - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 1024, - "height": 1024, - "num_inference_steps": 8 - }, - "backend": "openai-chat-omni" + } } ] } diff --git a/tests/dfx/perf/tests/test_hunyuan_image_tp2_sp2.json b/tests/dfx/perf/tests/test_hunyuan_image_tp2_sp2.json index 8d833f77ec0..e49cd0543ca 100644 --- a/tests/dfx/perf/tests/test_hunyuan_image_tp2_sp2.json +++ b/tests/dfx/perf/tests/test_hunyuan_image_tp2_sp2.json @@ -15,6 +15,7 @@ "local_model" ], "description": "TP=2 SP=2 baseline", + "server_type": "vllm-omni", "server_params": { "model": "tencent/HunyuanImage-3.0-Instruct", "serve_args": { @@ -29,27 +30,21 @@ "benchmark_params": [ { "name": "1024x1024_steps8", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "t2i", + "width": 1024, + "height": 1024, + "num-inference-steps": 8, + "num-prompts": 10, + "max-concurrency": 1, "baseline": { "H200": { - "request_throughput": 0.2, - "mean_peak_memory_mb": 97402, - "mean_e2el_ms": 5102.5 + "throughput_qps": 0.20, + "latency_p99": 5.1025, + "peak_memory_mb_max": 97402, + "peak_memory_mb_mean": 97402 } - }, - "dataset_name": "random", - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 1024, - "height": 1024, - "num_inference_steps": 8 - }, - "backend": "openai-chat-omni" + } } ] } diff --git a/tests/dfx/perf/tests/test_hunyuan_image_tp4.json b/tests/dfx/perf/tests/test_hunyuan_image_tp4.json index ae143092a96..54631471a38 100644 --- a/tests/dfx/perf/tests/test_hunyuan_image_tp4.json +++ b/tests/dfx/perf/tests/test_hunyuan_image_tp4.json @@ -15,6 +15,7 @@ "local_model" ], "description": "TP=4 baseline", + "server_type": "vllm-omni", "server_params": { "model": "tencent/HunyuanImage-3.0-Instruct", "serve_args": { @@ -28,27 +29,22 @@ "benchmark_params": [ { "name": "1024x1024_steps8", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "t2i", + "width": 1024, + "height": 1024, + "num-inference-steps": 8, + "num-prompts": 10, + "max-concurrency": 1, "baseline": { "H200": { - "request_throughput": 0.213, - "mean_e2el_ms": 4695.3, - "mean_peak_memory_mb": 56912.0 + "throughput_qps": 0.213, + "latency_mean": 4.6953, + "latency_p99": 4.8601, + "peak_memory_mb_max": 56912.0, + "peak_memory_mb_mean": 56912.0 } - }, - "dataset_name": "random", - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 1024, - "height": 1024, - "num_inference_steps": 8 - }, - "backend": "openai-chat-omni" + } } ] } diff --git a/tests/dfx/perf/tests/test_qwen_image_edit_2511_vllm_omni.json b/tests/dfx/perf/tests/test_qwen_image_edit_2511_vllm_omni.json index f9d4c921956..34d275ac971 100644 --- a/tests/dfx/perf/tests/test_qwen_image_edit_2511_vllm_omni.json +++ b/tests/dfx/perf/tests/test_qwen_image_edit_2511_vllm_omni.json @@ -14,6 +14,7 @@ "diffusion" ], "description": "Single-device baseline (two input images)", + "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image-Edit-2511", "serve_args": { @@ -23,69 +24,43 @@ "benchmark_params": [ { "name": "512x512_steps20_i2i_2img", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "i2i", + "width": 512, + "height": 512, + "num-inference-steps": 20, + "num-prompts": 10, + "max-concurrency": 1, + "num-input-images": 2, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.0684, - "mean_e2el_ms": 14624.9, - "mean_peak_memory_mb": 57632.0 + "throughput_qps": 0.0684, + "latency_mean": 14.6249, + "peak_memory_mb_max": 57632.0, + "peak_memory_mb_mean": 57632.0 } - }, - "dataset_name": "random-mm", - "random_mm_base_items_per_request": 2, - "random_mm_num_mm_items_range_ratio": 0, - "random_mm_limit_mm_per_prompt": { - "image": 2 - }, - "random_mm_bucket_config": { - "(512, 512, 1)": 1.0 - }, - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 512, - "height": 512, - "num_inference_steps": 20, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } }, { "name": "1536x1536_steps35_i2i_2img", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "i2i", + "width": 1536, + "height": 1536, + "num-inference-steps": 35, + "num-prompts": 10, + "max-concurrency": 1, + "num-input-images": 2, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.0177, - "mean_e2el_ms": 56528.5, - "mean_peak_memory_mb": 67840.0 + "throughput_qps": 0.0177, + "latency_mean": 56.5285, + "peak_memory_mb_max": 67840.0, + "peak_memory_mb_mean": 67840.0 } - }, - "dataset_name": "random-mm", - "random_mm_base_items_per_request": 2, - "random_mm_num_mm_items_range_ratio": 0, - "random_mm_limit_mm_per_prompt": { - "image": 2 - }, - "random_mm_bucket_config": { - "(512, 512, 1)": 1.0 - }, - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 1536, - "height": 1536, - "num_inference_steps": 35, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } } ] }, @@ -105,6 +80,7 @@ "diffusion" ], "description": "Ulysses SP=2 + CFG=2 + VAE patch parallel=4", + "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image-Edit-2511", "serve_args": { @@ -118,36 +94,23 @@ "benchmark_params": [ { "name": "1536x1536_steps35_i2i_2img", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "i2i", + "width": 1536, + "height": 1536, + "num-inference-steps": 35, + "num-prompts": 10, + "max-concurrency": 1, + "num-input-images": 2, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.0515, - "mean_e2el_ms": 19434.5, - "mean_peak_memory_mb": 56348.0 + "throughput_qps": 0.0515, + "latency_mean": 19.4345, + "peak_memory_mb_max": 56348.0, + "peak_memory_mb_mean": 56348.0 } - }, - "dataset_name": "random-mm", - "random_mm_base_items_per_request": 2, - "random_mm_num_mm_items_range_ratio": 0, - "random_mm_limit_mm_per_prompt": { - "image": 2 - }, - "random_mm_bucket_config": { - "(512, 512, 1)": 1.0 - }, - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 1536, - "height": 1536, - "num_inference_steps": 35, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } } ] }, @@ -166,6 +129,7 @@ "diffusion" ], "description": "Ulysses SP=2 + CFG=2 + CacheDiT", + "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image-Edit-2511", "serve_args": { @@ -189,69 +153,43 @@ "benchmark_params": [ { "name": "512x512_steps20_i2i_2img", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "i2i", + "width": 512, + "height": 512, + "num-inference-steps": 20, + "num-prompts": 10, + "max-concurrency": 1, + "num-input-images": 2, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.2479, - "mean_e2el_ms": 4038.9, - "mean_peak_memory_mb": 57750.0 + "throughput_qps": 0.2479, + "latency_mean": 4.0389, + "peak_memory_mb_max": 57750.0, + "peak_memory_mb_mean": 57750.0 } - }, - "dataset_name": "random-mm", - "random_mm_base_items_per_request": 2, - "random_mm_num_mm_items_range_ratio": 0, - "random_mm_limit_mm_per_prompt": { - "image": 2 - }, - "random_mm_bucket_config": { - "(512, 512, 1)": 1.0 - }, - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 512, - "height": 512, - "num_inference_steps": 20, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } }, { "name": "1536x1536_steps35_i2i_2img", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "i2i", + "width": 1536, + "height": 1536, + "num-inference-steps": 35, + "num-prompts": 10, + "max-concurrency": 1, + "num-input-images": 2, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.0948, - "mean_e2el_ms": 10546.2, - "mean_peak_memory_mb": 67764.0 + "throughput_qps": 0.0948, + "latency_mean": 10.5462, + "peak_memory_mb_max": 67764.0, + "peak_memory_mb_mean": 67764.0 } - }, - "dataset_name": "random-mm", - "random_mm_base_items_per_request": 2, - "random_mm_num_mm_items_range_ratio": 0, - "random_mm_limit_mm_per_prompt": { - "image": 2 - }, - "random_mm_bucket_config": { - "(512, 512, 1)": 1.0 - }, - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 1536, - "height": 1536, - "num_inference_steps": 35, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } } ] } diff --git a/tests/dfx/perf/tests/test_qwen_image_layered_vllm_omni.json b/tests/dfx/perf/tests/test_qwen_image_layered_vllm_omni.json index 19f02d403c0..bc0f5c9239f 100644 --- a/tests/dfx/perf/tests/test_qwen_image_layered_vllm_omni.json +++ b/tests/dfx/perf/tests/test_qwen_image_layered_vllm_omni.json @@ -14,6 +14,7 @@ "diffusion" ], "description": "Single-device baseline", + "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image-Layered", "serve_args": { @@ -23,69 +24,41 @@ "benchmark_params": [ { "name": "640x640_steps20_i2i", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "i2i", + "width": 640, + "height": 640, + "num-inference-steps": 20, + "num-prompts": 10, + "max-concurrency": 1, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.0703, - "mean_e2el_ms": 14208.0, - "mean_peak_memory_mb": 63225.1429 + "throughput_qps": 0.0703, + "latency_mean": 14.208, + "peak_memory_mb_max": 63225.1429, + "peak_memory_mb_mean": 63225.1429 } - }, - "dataset_name": "random-mm", - "random_mm_base_items_per_request": 1, - "random_mm_num_mm_items_range_ratio": 0, - "random_mm_limit_mm_per_prompt": { - "image": 1 - }, - "random_mm_bucket_config": { - "(512, 512, 1)": 1.0 - }, - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 640, - "height": 640, - "num_inference_steps": 20, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } }, { "name": "1024x1024_steps35_i2i", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "i2i", + "width": 1024, + "height": 1024, + "num-inference-steps": 35, + "num-prompts": 10, + "max-concurrency": 1, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.0417, - "mean_e2el_ms": 23914.0, - "mean_peak_memory_mb": 62912.0 + "throughput_qps": 0.0417, + "latency_mean": 23.914, + "peak_memory_mb_max": 62912.0, + "peak_memory_mb_mean": 62912.0 } - }, - "dataset_name": "random-mm", - "random_mm_base_items_per_request": 1, - "random_mm_num_mm_items_range_ratio": 0, - "random_mm_limit_mm_per_prompt": { - "image": 1 - }, - "random_mm_bucket_config": { - "(512, 512, 1)": 1.0 - }, - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 1024, - "height": 1024, - "num_inference_steps": 35, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } } ] } diff --git a/tests/dfx/perf/tests/test_qwen_image_vllm_omni.json b/tests/dfx/perf/tests/test_qwen_image_vllm_omni.json index 237cc3c5b6f..61cb334eb7b 100644 --- a/tests/dfx/perf/tests/test_qwen_image_vllm_omni.json +++ b/tests/dfx/perf/tests/test_qwen_image_vllm_omni.json @@ -5,10 +5,7 @@ { "hardware_marks": { "res": { - "cuda": [ - "H100", - "B200" - ] + "cuda": ["H100", "B200"] }, "num_cards": 1 } @@ -17,6 +14,7 @@ "diffusion" ], "description": "Single-device baseline (no parallelism)", + "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image", "serve_args": { @@ -26,53 +24,39 @@ "benchmark_params": [ { "name": "512x512_steps20", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "t2i", + "width": 512, + "height": 512, + "num-inference-steps": 20, + "num-prompts": 10, + "max-concurrency": 1, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.4153, - "mean_e2el_ms": 2406.5, - "mean_peak_memory_mb": 55062.0 + "throughput_qps": 0.4153, + "latency_mean": 2.4065, + "peak_memory_mb_mean": 55062.0 } - }, - "dataset_name": "random", - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 512, - "height": 512, - "num_inference_steps": 20, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } }, { "name": "1536x1536_steps35", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "t2i", + "width": 1536, + "height": 1536, + "num-inference-steps": 35, + "num-prompts": 10, + "max-concurrency": 1, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.0423, - "mean_e2el_ms": 23599.1, - "mean_peak_memory_mb": 64546.0 + "throughput_qps": 0.0423, + "latency_mean": 23.5991, + "peak_memory_mb_mean": 64546.0 } - }, - "dataset_name": "random", - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 1536, - "height": 1536, - "num_inference_steps": 35, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } } ] }, @@ -82,10 +66,7 @@ { "hardware_marks": { "res": { - "cuda": [ - "H100", - "B200" - ] + "cuda": ["H100", "B200"] }, "num_cards": 1 } @@ -94,6 +75,7 @@ "diffusion" ], "description": "Single-device step execution: sequential 512/1536 plus 512 concurrency sweep. One server (max-num-seqs 8).", + "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image", "serve_args": { @@ -105,104 +87,60 @@ "benchmark_params": [ { "name": "512x512_steps20", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "t2i", + "width": 512, + "height": 512, + "num-inference-steps": 20, + "num-prompts": 10, + "max-concurrency": 1, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.4175, - "mean_e2el_ms": 2395.3, - "mean_peak_memory_mb": 54952.0 + "throughput_qps": 0.4175, + "latency_mean": 2.3953, + "peak_memory_mb_mean": 54952.0 } - }, - "dataset_name": "random", - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 512, - "height": 512, - "num_inference_steps": 20, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } }, { "name": "1536x1536_steps35", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "t2i", + "width": 1536, + "height": 1536, + "num-inference-steps": 35, + "num-prompts": 10, + "max-concurrency": 1, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.0423, - "mean_e2el_ms": 23576.8, - "mean_peak_memory_mb": 64500.0 + "throughput_qps": 0.0423, + "latency_mean": 23.5768, + "peak_memory_mb_mean": 64500.0 } - }, - "dataset_name": "random", - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 1536, - "height": 1536, - "num_inference_steps": 35, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } }, { "name": "512x512_steps20_high_concurrency", - "num_prompts": [ - 20, - 40, - 80, - 160 - ], - "max_concurrency": [ - 1, - 2, - 4, - 8 - ], + "dataset": "random", + "task": "t2i", + "width": 512, + "height": 512, + "num-inference-steps": 20, + "num-prompts": [20, 40, 80, 160], + "max-concurrency": [1, 2, 4, 8], + "warmup-requests": 8, + "warmup-concurrency": 8, + "warmup-num-inference-steps": 20, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": [ - 0.3937, - 0.6606, - 0.6188, - 0.7752 - ], - "mean_e2el_ms": [ - 2546.6, - 3023.5, - 6394.8, - 10194.4 - ], - "mean_peak_memory_mb": [ - 55149.0, - 55150.0, - 55153.1458, - 55156.325 - ] + "throughput_qps": [0.3937, 0.6606, 0.6188, 0.7752], + "latency_mean": [2.5466, 3.0235, 6.3948, 10.1944], + "peak_memory_mb_mean": [55149.0, 55150.0, 55153.1458, 55156.325] } - }, - "num_warmups": 8, - "dataset_name": "random", - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 512, - "height": 512, - "num_inference_steps": 20, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } } ] }, @@ -212,10 +150,7 @@ { "hardware_marks": { "res": { - "cuda": [ - "H100", - "B200" - ] + "cuda": ["H100", "B200"] }, "num_cards": 4 } @@ -224,6 +159,7 @@ "diffusion" ], "description": "Ulysses SP=2 + CFG-parallel=2 + VAE Patch Parallel=4", + "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image", "serve_args": { @@ -237,28 +173,21 @@ "benchmark_params": [ { "name": "1536x1536_steps35", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "t2i", + "width": 1536, + "height": 1536, + "num-inference-steps": 35, + "num-prompts": 10, + "max-concurrency": 1, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.1214, - "mean_e2el_ms": 8221.1, - "mean_peak_memory_mb": 54474.0 + "throughput_qps": 0.1214, + "latency_mean": 8.2211, + "peak_memory_mb_mean": 54474.0 } - }, - "dataset_name": "random", - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 1536, - "height": 1536, - "num_inference_steps": 35, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } } ] }, @@ -268,10 +197,7 @@ { "hardware_marks": { "res": { - "cuda": [ - "H100", - "B200" - ] + "cuda": ["H100", "B200"] }, "num_cards": 4 } @@ -280,6 +206,7 @@ "diffusion" ], "description": "Ulysses SP=2 + CFG-parallel=2 + CacheDiT acceleration", + "server_type": "vllm-omni", "server_params": { "model": "Qwen/Qwen-Image", "serve_args": { @@ -303,53 +230,39 @@ "benchmark_params": [ { "name": "512x512_steps20", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "t2i", + "width": 512, + "height": 512, + "num-inference-steps": 20, + "num-prompts": 10, + "max-concurrency": 1, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.559, - "mean_e2el_ms": 1794.0, - "mean_peak_memory_mb": 55174.0 + "throughput_qps": 0.559, + "latency_mean": 1.794, + "peak_memory_mb_mean": 55174.0 } - }, - "dataset_name": "random", - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 512, - "height": 512, - "num_inference_steps": 20, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } }, { "name": "1536x1536_steps35", - "num_prompts": 10, - "max_concurrency": 1, + "dataset": "random", + "task": "t2i", + "width": 1536, + "height": 1536, + "num-inference-steps": 35, + "num-prompts": 10, + "max-concurrency": 1, + "enable-negative-prompt": true, "baseline": { "H100": { - "request_throughput": 0.1918, - "mean_e2el_ms": 5202.2, - "mean_peak_memory_mb": 64750.0 + "throughput_qps": 0.1918, + "latency_mean": 5.2022, + "peak_memory_mb_mean": 64750.0 } - }, - "dataset_name": "random", - "endpoint": "/v1/chat/completions", - "tokenizer": "gpt2", - "random_input_len": 8, - "random_output_len": 1, - "percentile_metrics": "e2el", - "extra_body": { - "width": 1536, - "height": 1536, - "num_inference_steps": 35, - "negative_prompt": "Negative prompt for benchmarking diffusion models" - }, - "backend": "openai-chat-omni" + } } ] } diff --git a/vllm_omni/benchmarks/patch/patch.py b/vllm_omni/benchmarks/patch/patch.py index 6f608de8050..3854098a568 100644 --- a/vllm_omni/benchmarks/patch/patch.py +++ b/vllm_omni/benchmarks/patch/patch.py @@ -765,6 +765,7 @@ class MixRequestFuncOutput(RequestFuncOutput): "negative_prompt", "num_inference_steps", "guidance_scale", + "guidance_scale_2", "strength", "true_cfg_scale", "seed", diff --git a/vllm_omni/diffusion/models/cosmos3/pipeline_cosmos3.py b/vllm_omni/diffusion/models/cosmos3/pipeline_cosmos3.py index 04f25ac58d9..95ba4767fe9 100644 --- a/vllm_omni/diffusion/models/cosmos3/pipeline_cosmos3.py +++ b/vllm_omni/diffusion/models/cosmos3/pipeline_cosmos3.py @@ -1730,7 +1730,10 @@ def _forward_robolab_policy( }, }, } - return DiffusionOutput(output=action_output) + return DiffusionOutput( + output=action_output, + stage_durations=self.stage_durations if hasattr(self, "stage_durations") else None, + ) @staticmethod def _truthy(value) -> bool: @@ -3610,6 +3613,7 @@ def _forward_transfer( "payload": {"video": output_video}, "metadata": {"video": {"fps": frame_rate}}, }, + stage_durations=self.stage_durations if hasattr(self, "stage_durations") else None, ) full_output = torch.cat(output_chunks, dim=2)[:, :, :total_frames] @@ -3636,6 +3640,7 @@ def _forward_transfer( "video": {"fps": frame_rate}, }, }, + stage_durations=self.stage_durations if hasattr(self, "stage_durations") else None, ) # -- Forward (main generation entry point) ------------------------------- @@ -4072,7 +4077,10 @@ def _run_diffusion(start_latents): if _is_rank_zero(): logger.info("Sound tokenizer decoded in %.2fs", time.time() - sound_decode_start) logger.info("Total pipeline time: %.2fs", time.time() - pipeline_start) - return DiffusionOutput(output={"video": video, "audio": audio, "audio_sample_rate": sound_sample_rate}) + return DiffusionOutput( + output={"video": video, "audio": audio, "audio_sample_rate": sound_sample_rate}, + stage_durations=self.stage_durations if hasattr(self, "stage_durations") else None, + ) if action_enabled: if action_latents is None or raw_action_dim is None or domain_id is None: @@ -4092,6 +4100,10 @@ def _run_diffusion(start_latents): }, }, }, + stage_durations=self.stage_durations if hasattr(self, "stage_durations") else None, ) - return DiffusionOutput(output={"image": video} if is_t2i else {"video": video}) + return DiffusionOutput( + output={"image": video} if is_t2i else {"video": video}, + stage_durations=self.stage_durations if hasattr(self, "stage_durations") else None, + ) diff --git a/vllm_omni/diffusion/models/hunyuan_video/pipeline_hunyuan_video_1_5.py b/vllm_omni/diffusion/models/hunyuan_video/pipeline_hunyuan_video_1_5.py index 942e0b28435..5f0ecb452e1 100644 --- a/vllm_omni/diffusion/models/hunyuan_video/pipeline_hunyuan_video_1_5.py +++ b/vllm_omni/diffusion/models/hunyuan_video/pipeline_hunyuan_video_1_5.py @@ -1,5 +1,5 @@ # SPDX-License-Identifier: Apache-2.0 -# SPDX-FileCopyrightText: Copyright contributors to the vLLM project +# SPDX-FileCopyrightText: Copyright contributors to the vLLM-Omni project from __future__ import annotations @@ -548,7 +548,10 @@ def forward( latents = latents.to(self.vae.dtype) / self.vae.config.scaling_factor output = self.vae.decode(latents, return_dict=False)[0] - return DiffusionOutput(output=output) + return DiffusionOutput( + output=output, + stage_durations=self.stage_durations if hasattr(self, "stage_durations") else None, + ) def load_weights(self, weights: Iterable[tuple[str, torch.Tensor]]) -> set[str]: loader = AutoWeightsLoader(self) diff --git a/vllm_omni/diffusion/models/hunyuan_video/pipeline_hunyuan_video_1_5_i2v.py b/vllm_omni/diffusion/models/hunyuan_video/pipeline_hunyuan_video_1_5_i2v.py index 1281a9e39d6..f01d45a4ec9 100644 --- a/vllm_omni/diffusion/models/hunyuan_video/pipeline_hunyuan_video_1_5_i2v.py +++ b/vllm_omni/diffusion/models/hunyuan_video/pipeline_hunyuan_video_1_5_i2v.py @@ -1,5 +1,5 @@ # SPDX-License-Identifier: Apache-2.0 -# SPDX-FileCopyrightText: Copyright contributors to the vLLM project +# SPDX-FileCopyrightText: Copyright contributors to the vLLM-Omni project from __future__ import annotations @@ -643,7 +643,10 @@ def forward( latents = latents.to(self.vae.dtype) / self.vae.config.scaling_factor output = self.vae.decode(latents, return_dict=False)[0] - return DiffusionOutput(output=output) + return DiffusionOutput( + output=output, + stage_durations=self.stage_durations if hasattr(self, "stage_durations") else None, + ) def load_weights(self, weights: Iterable[tuple[str, torch.Tensor]]) -> set[str]: loader = AutoWeightsLoader(self) From d080c9c534a8593867ffcae125aaf5cbb2c3654b Mon Sep 17 00:00:00 2001 From: wangyu <410167048@qq.com> Date: Sat, 19 Sep 2026 16:44:46 +0800 Subject: [PATCH 08/15] Enhance stage duration metrics handling in benchmarks - Updated `aggregate_stage_durations` to exclude `stage_N_gen_ms` from the metrics, ensuring clarity in the reported timings. - Modified `print_stage_durations_metrics` to utilize metrics stored in `MultiModalsBenchmarkMetrics`, improving the output format and consistency. - Adjusted test cases to validate the exclusion of `stage_N_gen_ms` and ensure accurate assertions in metrics display. These changes improve the accuracy and readability of stage duration metrics during benchmarking, facilitating better performance analysis. Signed-off-by: wangyu <410167048@qq.com> --- tests/benchmarks/metrics/test_metrics.py | 45 ++++++++++++++-- vllm_omni/benchmarks/metrics/metrics.py | 67 +++++++++++++++++------- vllm_omni/benchmarks/patch/patch.py | 8 ++- vllm_omni/metrics/definitions.py | 4 ++ 4 files changed, 96 insertions(+), 28 deletions(-) diff --git a/tests/benchmarks/metrics/test_metrics.py b/tests/benchmarks/metrics/test_metrics.py index 6e572e63a80..e7d782f9e02 100644 --- a/tests/benchmarks/metrics/test_metrics.py +++ b/tests/benchmarks/metrics/test_metrics.py @@ -591,10 +591,10 @@ def test_image_with_generated_text_still_reports_text_result(capsys): def test_aggregate_stage_durations_mean_p50_p99() -> None: ok_a = MixRequestFuncOutput() ok_a.success = True - ok_a.stage_durations = {"diffuse": 1.0, "vae.decode": 0.2} + ok_a.stage_durations = {"diffuse": 1.0, "vae.decode": 0.2, "stage_0_gen_ms": 50.0} ok_b = MixRequestFuncOutput() ok_b.success = True - ok_b.stage_durations = {"diffuse": 3.0, "vae.decode": 0.4} + ok_b.stage_durations = {"diffuse": 3.0, "vae.decode": 0.4, "stage_1_gen_ms": 80.0} failed = MixRequestFuncOutput() failed.success = False failed.stage_durations = {"diffuse": 99.0} @@ -604,6 +604,8 @@ def test_aggregate_stage_durations_mean_p50_p99() -> None: assert summaries["stage_durations_p50"]["diffuse"] == pytest.approx(2.0) assert summaries["stage_durations_mean"]["vae.decode"] == pytest.approx(0.3) assert "stage_durations_p99" in summaries + assert "stage_0_gen_ms" not in summaries["stage_durations_mean"] + assert "stage_1_gen_ms" not in summaries["stage_durations_mean"] def test_print_stage_durations_metrics(capsys) -> None: @@ -613,8 +615,11 @@ def test_print_stage_durations_metrics(capsys) -> None: "Wan22I2VPipeline.diffuse": 1.25, "Wan22I2VPipeline.text_encoder.forward": 0.4, "queue_wait_ms": 0.5, + "stage_0_gen_ms": 1000.0, } - print_stage_durations_metrics([output]) + summaries = aggregate_stage_durations([output]) + metrics = SimpleNamespace(**summaries) + print_stage_durations_metrics([99.0], metrics) out = capsys.readouterr().out assert "Wan22I2VPipeline" not in out assert "Mean Diffuse (s):" in out @@ -622,6 +627,40 @@ def test_print_stage_durations_metrics(capsys) -> None: assert "P99 Diffuse (s):" in out assert "Mean Text Encoder Forward (s):" in out assert "Mean Queue Wait (ms):" in out + assert "Stage 0 Gen" not in out + assert "stage_0_gen" not in out + + +def test_profiler_stage_durations_print_without_print_stage(capsys) -> None: + output = MixRequestFuncOutput() + output.success = True + output.prompt_len = 8 + output.latency = 1.0 + output.stage_durations = { + "Wan22I2VPipeline.diffuse": 1.25, + "queue_wait_ms": 0.5, + "stage_0_gen_ms": 1000.0, + } + common = dict( + input_requests=[], + outputs=[output], + dur_s=1.0, + tokenizer=None, + selected_percentiles=[50.0, 99.0], + goodput_config_dict={}, + task_type=TaskType.GENERATION, + selected_percentile_metrics=["e2el"], + max_concurrency=None, + request_rate=float("inf"), + benchmark_duration=1.0, + ) + metrics, _ = calculate_metrics(**common, print_stage=False) + shown = capsys.readouterr().out + assert "Mean Diffuse (s):" in shown + assert "Mean Queue Wait (ms):" in shown + assert "Stage 0 Gen" not in shown + assert metrics.stage_durations_mean["Wan22I2VPipeline.diffuse"] == pytest.approx(1.25) + assert "stage_0_gen_ms" not in metrics.stage_durations_mean if __name__ == "__main__": diff --git a/vllm_omni/benchmarks/metrics/metrics.py b/vllm_omni/benchmarks/metrics/metrics.py index f71cb291d1c..6cfe473d5b1 100644 --- a/vllm_omni/benchmarks/metrics/metrics.py +++ b/vllm_omni/benchmarks/metrics/metrics.py @@ -16,6 +16,7 @@ from vllm_omni.metrics import definitions as defs _PERCENTILE_ROWS_TYPE = list[tuple[float, float]] | None +_STAGE_DURATION_MAP_TYPE = dict[str, float] | None _FLOAT_LIST_TYPE = list[float] _INT_LIST_TYPE = list[int] @@ -68,6 +69,9 @@ (defs.MEDIAN_PEAK_MEMORY_MB, float, field(default=0.0)), (defs.STD_PEAK_MEMORY_MB, float, field(default=0.0)), (defs.PERCENTILES_PEAK_MEMORY_MB, _PERCENTILE_ROWS_TYPE, field(default=None)), + (defs.STAGE_DURATIONS_MEAN, _STAGE_DURATION_MAP_TYPE, field(default=None)), + (defs.STAGE_DURATIONS_P50, _STAGE_DURATION_MAP_TYPE, field(default=None)), + (defs.STAGE_DURATIONS_P99, _STAGE_DURATION_MAP_TYPE, field(default=None)), ] # ``make_dataclass`` returns a runtime class that mypy treats as a variable, not @@ -243,6 +247,7 @@ def print_metrics( if isinstance(metrics, MultiModalsBenchmarkMetrics): print("{:<40} {:<10.2f}".format("Peak concurrent requests:", metrics.max_concurrent_requests)) print_peak_memory_metrics(metrics) + print_stage_durations_metrics(selected_percentiles or [], metrics) if task_type != TaskType.GENERATION or "e2el" in selected_percentile_metrics: process_one_metric("e2el", metrics) print_text_metrics(task_type, selected_percentile_metrics, metrics) @@ -253,7 +258,6 @@ def print_metrics( print_image_metrics(selected_percentiles or [], metrics) if _has_video_output(metrics): print_video_metrics(selected_percentiles or [], metrics) - print_stage_durations_metrics(outputs) if print_stage and outputs and selected_percentiles is not None: stage_metrics = _build_stage_metrics_from_outputs(outputs) if stage_metrics: @@ -343,10 +347,12 @@ def print_peak_memory_metrics(metrics: MultiModalsBenchmarkMetrics): def aggregate_stage_durations(outputs: Sequence[RequestFuncOutput]) -> dict[str, dict[str, float]]: - """Aggregate per-request pipeline profiler timings into mean/p50/p99 maps. + """Aggregate pipeline profiler timings into mean/p50/p99 maps. - Mirrors ``diffusion_benchmark_serving`` so ``--save-result`` JSON can carry - ``stage_durations_{mean,p50,p99}`` for Diffuse / VAE / TextEncoder keys. + ``stage_N_gen_ms`` is omitted. That clock already belongs to + ``stage_gen_time`` / image / video generation, not to this profiler block. + The maps are stored on ``MultiModalsBenchmarkMetrics`` and printed with the + other run-level metrics. ``--print-stage`` does not gate them. """ stage_duration_lists: dict[str, list[float]] = {} for output in outputs: @@ -356,17 +362,19 @@ def aggregate_stage_durations(outputs: Sequence[RequestFuncOutput]) -> dict[str, if not isinstance(stage_durations, dict): continue for stage, duration in stage_durations.items(): + if _is_printed_stage_gen_key(str(stage)): + continue if isinstance(duration, bool) or not isinstance(duration, (int, float)) or not np.isfinite(duration): continue stage_duration_lists.setdefault(str(stage), []).append(float(duration)) if not stage_duration_lists: return {} return { - "stage_durations_mean": {stage: float(np.mean(values)) for stage, values in stage_duration_lists.items()}, - "stage_durations_p50": { + defs.STAGE_DURATIONS_MEAN: {stage: float(np.mean(values)) for stage, values in stage_duration_lists.items()}, + defs.STAGE_DURATIONS_P50: { stage: float(np.percentile(values, 50)) for stage, values in stage_duration_lists.items() }, - "stage_durations_p99": { + defs.STAGE_DURATIONS_P99: { stage: float(np.percentile(values, 99)) for stage, values in stage_duration_lists.items() }, } @@ -377,7 +385,7 @@ def _stage_duration_display_name(stage: str) -> str: ``Wan22I2VPipeline.diffuse`` -> ``Diffuse`` ``Wan22I2VPipeline.text_encoder.forward`` -> ``Text Encoder Forward`` - ``queue_wait_ms`` / ``stage_0_gen_ms`` -> ``Queue Wait`` / ``Stage 0 Gen`` + ``queue_wait_ms`` -> ``Queue Wait`` """ name = str(stage) head, sep, tail = name.partition(".") @@ -388,25 +396,40 @@ def _stage_duration_display_name(stage: str) -> str: return name.replace("_", " ").replace(".", " ").title() -def print_stage_durations_metrics(outputs: Sequence[RequestFuncOutput] | None) -> None: - """Print per-stage mean / median / p99 in the same style as other bench metrics.""" - if not outputs: - return - summaries = aggregate_stage_durations(outputs) - mean = summaries.get("stage_durations_mean") or {} +def _is_printed_stage_gen_key(stage: str) -> bool: + """True for ``stage_0_gen_ms``, which duplicates ``--print-stage`` stage timing.""" + name = str(stage) + prefix, suffix = "stage_", "_gen_ms" + if not name.startswith(prefix) or not name.endswith(suffix): + return False + return name[len(prefix) : -len(suffix)].isdigit() + + +def print_stage_durations_metrics( + selected_percentiles: list[float], + metrics: MultiModalsBenchmarkMetrics, +) -> None: + """Print profiler timings already stored on ``metrics``. + + Mean and median always print, matching :func:`print_image_metrics`. P99 prints + only when 99 is in ``selected_percentiles``. ``stage_N_gen_ms`` is not here; + that clock stays on stage_gen_time / image / video generation. + """ + mean = getattr(metrics, defs.STAGE_DURATIONS_MEAN, None) or {} if not mean: return - p50 = summaries.get("stage_durations_p50") or {} - p99 = summaries.get("stage_durations_p99") or {} + p50 = getattr(metrics, defs.STAGE_DURATIONS_P50, None) or {} + p99 = getattr(metrics, defs.STAGE_DURATIONS_P99, None) or {} + show_p99 = any(float(p) == 99.0 for p in selected_percentiles) print("{s:{c}^{n}}".format(s=" Stage Durations ", n=50, c="-")) for stage in mean: label = _stage_duration_display_name(stage) unit = " (ms)" if str(stage).endswith("_ms") else " (s)" - print("{:<40} {:<10.4f}".format(f"Mean {label}{unit}:", mean[stage])) + print("{:<40} {:<10.2f}".format(f"Mean {label}{unit}:", mean[stage])) if stage in p50: - print("{:<40} {:<10.4f}".format(f"Median {label}{unit}:", p50[stage])) - if stage in p99: - print("{:<40} {:<10.4f}".format(f"P99 {label}{unit}:", p99[stage])) + print("{:<40} {:<10.2f}".format(f"Median {label}{unit}:", p50[stage])) + if show_p99 and stage in p99: + print("{:<40} {:<10.2f}".format(f"P99 {label}{unit}:", p99[stage])) def print_image_metrics(selected_percentiles: list[float], metrics: MultiModalsBenchmarkMetrics): @@ -1153,6 +1176,7 @@ def _formatwarning( isinstance(getattr(output, "duplex_session_metrics", None), dict) for output in outputs ) missing_duplex_value = float("nan") if duplex_metrics_present else 0 + stage_duration_summaries = aggregate_stage_durations(outputs) metrics = MultiModalsBenchmarkMetrics( completed=completed, failed=len(failed_outputs), @@ -1246,6 +1270,9 @@ def _formatwarning( defs.AUDIO_CONTINUITY_OK_RATE: ( (sum(audio_continuity_ok) / len(audio_continuity_ok)) if audio_continuity_ok else 1.0 ), + defs.STAGE_DURATIONS_MEAN: stage_duration_summaries.get(defs.STAGE_DURATIONS_MEAN) or {}, + defs.STAGE_DURATIONS_P50: stage_duration_summaries.get(defs.STAGE_DURATIONS_P50) or {}, + defs.STAGE_DURATIONS_P99: stage_duration_summaries.get(defs.STAGE_DURATIONS_P99) or {}, }, ) print_metrics( diff --git a/vllm_omni/benchmarks/patch/patch.py b/vllm_omni/benchmarks/patch/patch.py index 3854098a568..acfa69feaff 100644 --- a/vllm_omni/benchmarks/patch/patch.py +++ b/vllm_omni/benchmarks/patch/patch.py @@ -2692,7 +2692,6 @@ async def async_request_openai_realtime_duplex( from vllm_omni.benchmarks.metrics.metrics import ( MultiModalsBenchmarkMetrics, - aggregate_stage_durations, calculate_metrics, has_metric_samples, ) @@ -3047,6 +3046,9 @@ def measured_ttft(output: RequestFuncOutput) -> float | None: defs.MEAN_PEAK_MEMORY_MB: getattr(metrics, defs.MEAN_PEAK_MEMORY_MB), defs.MEDIAN_PEAK_MEMORY_MB: getattr(metrics, defs.MEDIAN_PEAK_MEMORY_MB), defs.PERCENTILES_PEAK_MEMORY_MB: getattr(metrics, defs.PERCENTILES_PEAK_MEMORY_MB), + defs.STAGE_DURATIONS_MEAN: getattr(metrics, defs.STAGE_DURATIONS_MEAN) or {}, + defs.STAGE_DURATIONS_P50: getattr(metrics, defs.STAGE_DURATIONS_P50) or {}, + defs.STAGE_DURATIONS_P99: getattr(metrics, defs.STAGE_DURATIONS_P99) or {}, "input_lens": [output.prompt_len for output in outputs], "start_times": [output.start_time for output in outputs], "output_lens": actual_output_lens, @@ -3107,10 +3109,6 @@ def measured_ttft(output: RequestFuncOutput) -> float | None: if omniinteract_summary is not None: result["omniinteract"] = omniinteract_summary - stage_duration_summaries = aggregate_stage_durations(outputs) - if stage_duration_summaries: - result.update(stage_duration_summaries) - from vllm_omni.benchmarks.data_modules.daily_omni_eval import ( compute_daily_omni_accuracy_metrics, print_daily_omni_accuracy_summary, diff --git a/vllm_omni/metrics/definitions.py b/vllm_omni/metrics/definitions.py index a09fbcc2ff2..59b575e50fc 100644 --- a/vllm_omni/metrics/definitions.py +++ b/vllm_omni/metrics/definitions.py @@ -99,6 +99,10 @@ STD_PEAK_MEMORY_MB = f"std_{PEAK_MEMORY_MB}" PERCENTILES_PEAK_MEMORY_MB = f"percentiles_{PEAK_MEMORY_MB}" +STAGE_DURATIONS_MEAN = "stage_durations_mean" +STAGE_DURATIONS_P50 = "stage_durations_p50" +STAGE_DURATIONS_P99 = "stage_durations_p99" + # Stage snapshot / StageBenchmarkMetrics field names. TOTAL_OUTPUT = "total_output" TTFTS = "ttfts" From 6b379c1b408610a6aa38ce3ce59c46ced67b983d Mon Sep 17 00:00:00 2001 From: wangyu <410167048@qq.com> Date: Sun, 20 Sep 2026 00:14:31 +0800 Subject: [PATCH 09/15] Add video reference handling to benchmarks - Introduced `_iter_video_reference_inputs` to yield video references from multimodal content. - Updated `_add_video_reference_to_form` to handle structured video references and integrate them into form data. - Enhanced tests to validate the extraction and addition of video references, ensuring compatibility with the existing image reference handling. These changes improve the support for video content in the benchmarking framework, aligning with the existing image reference functionality. Signed-off-by: wangyu <410167048@qq.com> --- tests/benchmarks/patch/test_patch.py | 29 ++++++++++ vllm_omni/benchmarks/patch/patch.py | 83 ++++++++++++++++++++++++---- 2 files changed, 102 insertions(+), 10 deletions(-) diff --git a/tests/benchmarks/patch/test_patch.py b/tests/benchmarks/patch/test_patch.py index 2121b100d69..2eebb435fe9 100644 --- a/tests/benchmarks/patch/test_patch.py +++ b/tests/benchmarks/patch/test_patch.py @@ -30,6 +30,7 @@ _attach_seed_tts_to_request_func_input, _build_benchmark_session, _extract_stage_durations_from_payload, + _iter_video_reference_inputs, _omni_request_timeout_s, async_request_openai_chat_omni_completions, async_request_openai_image_edits_omni, @@ -1546,6 +1547,34 @@ def tracking_add_field(self, name, value=None, **kwargs): assert json.loads(payload) == reference +def test_video_reference_urls_from_random_mm_content(mocker: MockerFixture) -> None: + """random-mm video_url parts use the same form helper as image_reference.""" + import aiohttp + + content = [ + { + "type": "video_url", + "video_url": {"url": "data:video/mp4;base64,AAAA"}, + } + ] + urls = list(_iter_video_reference_inputs(content)) + assert urls == ["data:video/mp4;base64,AAAA"] + + captured: list[tuple[str, object]] = [] + real_add_field = aiohttp.FormData.add_field + + def tracking_add_field(self, name, value=None, **kwargs): + captured.append((str(name), value)) + return real_add_field(self, name, value, **kwargs) + + mocker.patch.object(aiohttp.FormData, "add_field", tracking_add_field) + form = aiohttp.FormData() + assert _add_video_reference_to_form(form, urls[0]) is True + payload = next(value for name, value in captured if name == "video_reference") + assert isinstance(payload, (str, bytes, bytearray)) + assert json.loads(payload) == {"video_url": "data:video/mp4;base64,AAAA"} + + def test_video_unsupported_image_reference_raises() -> None: import aiohttp diff --git a/vllm_omni/benchmarks/patch/patch.py b/vllm_omni/benchmarks/patch/patch.py index 505011b2b50..34910259fd3 100644 --- a/vllm_omni/benchmarks/patch/patch.py +++ b/vllm_omni/benchmarks/patch/patch.py @@ -785,13 +785,13 @@ def _guess_mime_type(path: str) -> str: return mime or "application/octet-stream" -def _iter_image_edit_inputs(value: Any) -> Iterable[Any]: +def _iter_image_reference_inputs(value: Any) -> Iterable[Any]: """Yield image references from benchmark multimodal content.""" if value is None: return if isinstance(value, list): for item in value: - yield from _iter_image_edit_inputs(item) + yield from _iter_image_reference_inputs(item) return if not isinstance(value, dict): yield value @@ -810,7 +810,38 @@ def _iter_image_edit_inputs(value: Any) -> Iterable[Any]: for key in ("image", "images"): if key in value: - yield from _iter_image_edit_inputs(value[key]) + yield from _iter_image_reference_inputs(value[key]) + + +def _iter_video_reference_inputs(value: Any) -> Iterable[str]: + """Yield video references from benchmark multimodal content. + + ``random-mm`` video buckets arrive as OpenAI chat parts + ``{"type": "video_url", "video_url": {"url": ...}}``. The videos API + expects ``video_reference`` with a string ``video_url``. + """ + if value is None: + return + if isinstance(value, list): + for item in value: + yield from _iter_video_reference_inputs(item) + return + if not isinstance(value, dict): + return + + if value.get("type") == "video_url": + video_url = value.get("video_url") + if isinstance(video_url, dict): + url = video_url.get("url") + if isinstance(url, str) and url: + yield url + elif isinstance(video_url, str) and video_url: + yield video_url + return + + for key in ("video", "videos"): + if key in value: + yield from _iter_video_reference_inputs(value[key]) def _add_image_edit_input_to_form(form: aiohttp.FormData, image_input: Any) -> None: @@ -1275,6 +1306,12 @@ def _is_structured_image_reference(reference: Mapping[str, object]) -> bool: return has_url or has_file_id +def _is_structured_video_reference(reference: Mapping[str, object]) -> bool: + """True for API video_reference objects ({"video_url": "..."}).""" + video_url = reference.get("video_url") + return isinstance(video_url, str) and bool(video_url) + + def _add_video_reference_to_form(form: aiohttp.FormData, reference: object) -> bool: if isinstance(reference, dict) and "bytes" in reference: form.add_field( @@ -1289,16 +1326,26 @@ def _add_video_reference_to_form(form: aiohttp.FormData, reference: object) -> b form.add_field("image_reference", json.dumps(dict(reference))) return True + if isinstance(reference, Mapping) and _is_structured_video_reference(reference): + form.add_field("video_reference", json.dumps(dict(reference))) + return True + if isinstance(reference, list): if reference and all(isinstance(item, Mapping) and _is_structured_image_reference(item) for item in reference): form.add_field("image_reference", json.dumps([dict(item) for item in reference])) return True + if reference and all(isinstance(item, Mapping) and _is_structured_video_reference(item) for item in reference): + form.add_field("video_reference", json.dumps([dict(item) for item in reference])) + return True raise ValueError( "Unsupported image_reference list; expected non-empty list of " '{"image_url": "..."} and/or {"file_id": "..."} objects.' ) if isinstance(reference, str): + if reference.startswith("data:video"): + form.add_field("video_reference", json.dumps({"video_url": reference})) + return True if reference.startswith(("data:image", "http://", "https://")): form.add_field("image_reference", json.dumps({"image_url": reference})) return True @@ -1344,8 +1391,9 @@ def _add_video_extra_body_to_form( "height", "poll_interval_s", "poll_timeout_s", - # Handled only by _add_video_reference_to_form (upload / JSON image_url). + # Handled only by _add_video_reference_to_form (upload / JSON image_url / video_url). "image_reference", + "video_reference", "input_reference", *_VIDEO_FORM_FIELDS, } @@ -1875,16 +1923,26 @@ async def async_request_openai_videos_omni( form.add_field("size", str(request_body["size"])) _add_video_extra_body_to_form(form, extra_body, request_body) - reference_added = False - for reference in _iter_image_edit_inputs(request_func_input.multi_modal_content): + image_reference_added = False + for reference in _iter_image_reference_inputs(request_func_input.multi_modal_content): if _add_video_reference_to_form(form, reference): - reference_added = True + image_reference_added = True break - if not reference_added: + if not image_reference_added: image_reference = extra_body.get("image_reference") if image_reference is not None: _add_video_reference_to_form(form, image_reference) + video_reference_added = False + for reference in _iter_video_reference_inputs(request_func_input.multi_modal_content): + if _add_video_reference_to_form(form, reference): + video_reference_added = True + break + if not video_reference_added: + video_reference = extra_body.get("video_reference") + if video_reference is not None: + _add_video_reference_to_form(form, video_reference) + headers = { "Authorization": f"Bearer {os.environ.get('OPENAI_API_KEY')}", } @@ -1993,7 +2051,7 @@ async def async_request_openai_image_edits_omni( _add_image_edit_extra_body_to_form(form, extra_body) try: - image_inputs = list(_iter_image_edit_inputs(request_func_input.multi_modal_content)) + image_inputs = list(_iter_image_reference_inputs(request_func_input.multi_modal_content)) if not image_inputs: raise ValueError( "openai-image-edits-omni requires image multimodal content. " @@ -2712,7 +2770,12 @@ def turn_settled() -> bool: output.ttft = (metric_mean(session_metrics.get("ttft_ms")) or 0.0) / 1000.0 output.audio_ttfp = (metric_mean(session_metrics.get("ttfp_ms")) or 0.0) / 1000.0 output.audio_rtf = metric_mean(session_metrics.get("rtf")) or 0.0 - output.audio_duration = sum(float(metric.get("audio_duration_ms") or 0.0) for metric in turn_metrics) / 1000.0 + audio_duration_ms = 0.0 + for metric in turn_metrics: + duration_ms = metric.get("audio_duration_ms") + if isinstance(duration_ms, (int, float)) and not isinstance(duration_ms, bool): + audio_duration_ms += duration_ms + output.audio_duration = audio_duration_ms / 1000.0 output.audio_frames = int(output.audio_duration * _SEED_TTS_OUTPUT_SAMPLE_RATE_HZ) output.latency = request_finished_at - output.start_time output.tts_turn_pcm_bytes = turn_pcm_bytes From bf9e36bf41f4e7df232ebbd8d4fa5373153b1b6e Mon Sep 17 00:00:00 2001 From: wangyu <410167048@qq.com> Date: Sun, 20 Sep 2026 13:22:16 +0800 Subject: [PATCH 10/15] Refactor video reference handling in benchmarks - Updated `_add_video_reference_to_form` to handle inline video data URLs more effectively, ensuring they are added as binary input references instead of JSON strings. - Adjusted tests to validate the new handling of video references, ensuring that the correct fields are populated and that no legacy fields are present. - Modified benchmark configuration to reflect changes in video dimensions. These improvements enhance the robustness of video content handling in the benchmarking framework. Signed-off-by: wangyu <410167048@qq.com> --- tests/benchmarks/patch/test_patch.py | 7 +++--- .../perf/tests/test_minimax_h3_vllm_omni.json | 2 +- vllm_omni/benchmarks/patch/patch.py | 25 ++++++++++++++++++- 3 files changed, 29 insertions(+), 5 deletions(-) diff --git a/tests/benchmarks/patch/test_patch.py b/tests/benchmarks/patch/test_patch.py index 2eebb435fe9..6926e39d19d 100644 --- a/tests/benchmarks/patch/test_patch.py +++ b/tests/benchmarks/patch/test_patch.py @@ -1570,9 +1570,10 @@ def tracking_add_field(self, name, value=None, **kwargs): mocker.patch.object(aiohttp.FormData, "add_field", tracking_add_field) form = aiohttp.FormData() assert _add_video_reference_to_form(form, urls[0]) is True - payload = next(value for name, value in captured if name == "video_reference") - assert isinstance(payload, (str, bytes, bytearray)) - assert json.loads(payload) == {"video_url": "data:video/mp4;base64,AAAA"} + uploaded = next(value for name, value in captured if name == "input_references") + assert uploaded == base64.b64decode("AAAA") + assert "input_reference" not in [name for name, _ in captured] + assert "video_reference" not in [name for name, _ in captured] def test_video_unsupported_image_reference_raises() -> None: diff --git a/tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json b/tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json index fb64945356d..3bf68064eee 100644 --- a/tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json +++ b/tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json @@ -202,7 +202,7 @@ "video": 1 }, "random_mm_bucket_config": { - "(768, 1344, 209)": 1.0 + "(480, 832, 209)": 1.0 }, "extra_body": { "width": 1344, diff --git a/vllm_omni/benchmarks/patch/patch.py b/vllm_omni/benchmarks/patch/patch.py index 34910259fd3..b61c94c6db2 100644 --- a/vllm_omni/benchmarks/patch/patch.py +++ b/vllm_omni/benchmarks/patch/patch.py @@ -1327,6 +1327,10 @@ def _add_video_reference_to_form(form: aiohttp.FormData, reference: object) -> b return True if isinstance(reference, Mapping) and _is_structured_video_reference(reference): + video_url = reference.get("video_url") + # Inline data URLs are too large for a text form field (1MB part limit). + if isinstance(video_url, str) and video_url.startswith("data:video"): + return _add_video_reference_to_form(form, video_url) form.add_field("video_reference", json.dumps(dict(reference))) return True @@ -1344,7 +1348,25 @@ def _add_video_reference_to_form(form: aiohttp.FormData, reference: object) -> b if isinstance(reference, str): if reference.startswith("data:video"): - form.add_field("video_reference", json.dumps({"video_url": reference})) + header, _, payload = reference.partition(",") + if not payload: + raise ValueError(f"Unsupported video data URL: {reference[:64]!r}") + try: + video_bytes = base64.b64decode(payload) + except (ValueError, TypeError) as exc: + raise ValueError("video data URL is not valid base64") from exc + mime = header[len("data:") :].split(";", 1)[0] or "video/mp4" + suffix = ".mp4" if mime.endswith("mp4") else ".bin" + form.add_field( + # Plural field persists the container to disk. Singular + # ``input_reference`` would decode every frame in the API + # process and trip Starlette / MiniMax size limits on + # random-mm videos. + "input_references", + video_bytes, + filename=f"benchmark-reference{suffix}", + content_type=mime, + ) return True if reference.startswith(("data:image", "http://", "https://")): form.add_field("image_reference", json.dumps({"image_url": reference})) @@ -1395,6 +1417,7 @@ def _add_video_extra_body_to_form( "image_reference", "video_reference", "input_reference", + "input_references", *_VIDEO_FORM_FIELDS, } for key, value in extra_body.items(): From 8bedd93d1451964a8cc3d7ddb17b06b79767cd55 Mon Sep 17 00:00:00 2001 From: wangyu <410167048@qq.com> Date: Sun, 20 Sep 2026 13:35:10 +0800 Subject: [PATCH 11/15] debug for artifact_paths Signed-off-by: wangyu <410167048@qq.com> --- .buildkite/cuda/test-nightly.yml | 967 ++++++++++++++++--------------- 1 file changed, 484 insertions(+), 483 deletions(-) diff --git a/.buildkite/cuda/test-nightly.yml b/.buildkite/cuda/test-nightly.yml index a46c8f8bb86..c178dadb79d 100644 --- a/.buildkite/cuda/test-nightly.yml +++ b/.buildkite/cuda/test-nightly.yml @@ -8,113 +8,114 @@ env: HF_HUB_ETAG_TIMEOUT: 60 steps: + # TEMPORARY: only TTS Perf Test Single-GPU is active; all other jobs commented out. # Group: collapses under one heading in the Buildkite UI; child steps still run in parallel. - - group: ":card_index_dividers: Omni Model Test" - key: nightly-omni-test-group - depends_on: upload-nightly-pipeline - if: build.env("NIGHTLY") == "1" || build.pull_request.labels includes "nightly-test" - steps: - - - label: ":full_moon: Omni · Function Test with H100 · Single-GPU" - source_file_dependencies: - - omni_qwen3_omni_function - - omni_minicpmo_4_5_function - timeout_in_minutes: 120 - commands: - - pytest -sv tests/e2e --run-level "full_model" -m "full_model and H100 and B200 and omni and cards_1" --ignore=tests/e2e/accuracy --ignore=tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py - - - label: ":full_moon: Omni · Function Test with H100 · 2-GPU" - source_file_dependencies: - - omni_qwen3_omni_function - - omni_minicpmo_4_5_function - timeout_in_minutes: 90 - commands: - - pytest -sv tests/e2e --run-level "full_model" -m "full_model and H100 and B200 and omni and cards_2" --ignore=tests/e2e/accuracy --ignore=tests/e2e/online_serving/test_qwen3_omni_multi_replicas.py - - - label: ":full_moon: Omni · MiniCPM-o 4.5 Duplex Test" - source_file_dependencies: omni_minicpmo_4_5_duplex_function - timeout_in_minutes: 50 - commands: - - pytest -s -v tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py -m "full_model and cuda and H100 and B200 and omni and cards_1" --run-level "full_model" - - - label: ":full_moon: Omni · Doc Test with L4 · 4-GPU" - source_file_dependencies: omni_qwen2_5_omni_doc - timeout_in_minutes: 90 - commands: - - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" - - pytest -s -v tests/examples/ -m "full_model and omni and L4 and B200 and cards_4" --run-level "full_model" - - - label: ":full_moon: Omni · Doc Test with H100 · 2-GPU" - source_file_dependencies: omni_qwen3_omni_doc - timeout_in_minutes: 90 - commands: - - pytest -s -v tests/examples/ -m "full_model and omni and H100 and B200 and cards_2" --run-level "full_model" - - - label: ":full_moon: Omni · Accuracy Test" - source_file_dependencies: omni_qwen3_omni_accuracy - timeout_in_minutes: 180 - artifact_paths: - - tests/e2e/accuracy/qwen3_omni/results/qwen_omni_acc/*.json - commands: - - export SEED_TTS_WER_EVAL=1 - - export SEED_TTS_EVAL_DEVICE=cuda:1 - - pytest -s -v tests/e2e/accuracy/qwen3_omni/test_qwen3_omni.py -m "full_model and H100 and B200 and cards_2" --run-level full_model - - - label: ":full_moon: Omni · MiniCPM-o 4.5 · Accuracy Test" - source_file_dependencies: omni_minicpmo_4_5_accuracy - timeout_in_minutes: 180 - artifact_paths: - - tests/e2e/accuracy/minicpmo_4_5/results/*.json - commands: - - export SEED_TTS_WER_EVAL=1 - - export SEED_TTS_EVAL_DEVICE=cuda:1 - - pytest -s -v tests/e2e/accuracy/minicpmo_4_5/test_minicpmo_4_5.py -m "full_model and H100 and B200 and cuda and (cards_1 or cards_2)" --run-level full_model - - - label: ":full_moon: Omni · Perf Test · No Async Chunk" - source_file_dependencies: omni_qwen3_omni_perf - key: nightly-omni-performance-no-async-chunk - timeout_in_minutes: 300 - artifact_paths: - - tests/dfx/perf/results/*.json - commands: - - export BENCHMARK_DIR=tests/dfx/perf/results - - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_no_async_chunk.json -m "H100 and B200 and cards_2" - - - label: ":full_moon: Omni · Perf Test · Async Chunk" - source_file_dependencies: omni_qwen3_omni_perf - key: nightly-omni-performance-async-chunk - timeout_in_minutes: 180 - artifact_paths: - - tests/dfx/perf/results/*.json - commands: - - export BENCHMARK_DIR=tests/dfx/perf/results - - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_async_chunk.json -m "H100 and B200 and full_model and cards_2" - - - label: ":full_moon: Omni · MiniCPM-o 4.5 · Perf Test" - source_file_dependencies: omni_minicpmo_4_5_perf - key: nightly-omni-performance-minicpmo-4-5 - timeout_in_minutes: 180 - artifact_paths: - - tests/dfx/perf/results/*.json - commands: - - export BENCHMARK_DIR=tests/dfx/perf/results - - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minicpmo_4_5.json -m "H100 and B200 and cards_1" - - - label: ":full_moon: Omni · MiniCPM-o 4.5 · Duplex Seed-TTS Perf Test" - source_file_dependencies: omni_minicpmo_4_5_duplex_perf - key: nightly-omni-performance-minicpmo-4-5-duplex-seed-tts - timeout_in_minutes: 180 - artifact_paths: - - tests/dfx/perf/results/*.json - commands: - - export BENCHMARK_DIR=tests/dfx/perf/results - - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minicpmo_4_5_duplex_seed_tts.json -m "H100 and B200 and cards_1" - - - label: ":full_moon: Omni · Multi-Replica Startup Test with 4x H100" - source_file_dependencies: omni_qwen3_omni_function - timeout_in_minutes: 45 - commands: - - pytest -s -v tests/e2e/online_serving/test_qwen3_omni_multi_replicas.py -m "full_model and H100 and B200 and cards_4" --run-level "core_model" +# - group: ":card_index_dividers: Omni Model Test" +# key: nightly-omni-test-group +# depends_on: upload-nightly-pipeline +# if: build.env("NIGHTLY") == "1" || build.pull_request.labels includes "nightly-test" +# steps: + +# - label: ":full_moon: Omni · Function Test with H100 · Single-GPU" +# source_file_dependencies: +# - omni_qwen3_omni_function +# - omni_minicpmo_4_5_function +# timeout_in_minutes: 120 +# commands: +# - pytest -sv tests/e2e --run-level "full_model" -m "full_model and H100 and B200 and omni and cards_1" --ignore=tests/e2e/accuracy --ignore=tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py + +# - label: ":full_moon: Omni · Function Test with H100 · 2-GPU" +# source_file_dependencies: +# - omni_qwen3_omni_function +# - omni_minicpmo_4_5_function +# timeout_in_minutes: 90 +# commands: +# - pytest -sv tests/e2e --run-level "full_model" -m "full_model and H100 and B200 and omni and cards_2" --ignore=tests/e2e/accuracy --ignore=tests/e2e/online_serving/test_qwen3_omni_multi_replicas.py + +# - label: ":full_moon: Omni · MiniCPM-o 4.5 Duplex Test" +# source_file_dependencies: omni_minicpmo_4_5_duplex_function +# timeout_in_minutes: 50 +# commands: +# - pytest -s -v tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py -m "full_model and cuda and H100 and B200 and omni and cards_1" --run-level "full_model" + +# - label: ":full_moon: Omni · Doc Test with L4 · 4-GPU" +# source_file_dependencies: omni_qwen2_5_omni_doc +# timeout_in_minutes: 90 +# commands: +# - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" +# - pytest -s -v tests/examples/ -m "full_model and omni and L4 and B200 and cards_4" --run-level "full_model" + +# - label: ":full_moon: Omni · Doc Test with H100 · 2-GPU" +# source_file_dependencies: omni_qwen3_omni_doc +# timeout_in_minutes: 90 +# commands: +# - pytest -s -v tests/examples/ -m "full_model and omni and H100 and B200 and cards_2" --run-level "full_model" + +# - label: ":full_moon: Omni · Accuracy Test" +# source_file_dependencies: omni_qwen3_omni_accuracy +# timeout_in_minutes: 180 +# artifact_paths: +# - tests/e2e/accuracy/qwen3_omni/results/qwen_omni_acc/*.json +# commands: +# - export SEED_TTS_WER_EVAL=1 +# - export SEED_TTS_EVAL_DEVICE=cuda:1 +# - pytest -s -v tests/e2e/accuracy/qwen3_omni/test_qwen3_omni.py -m "full_model and H100 and B200 and cards_2" --run-level full_model + +# - label: ":full_moon: Omni · MiniCPM-o 4.5 · Accuracy Test" +# source_file_dependencies: omni_minicpmo_4_5_accuracy +# timeout_in_minutes: 180 +# artifact_paths: +# - tests/e2e/accuracy/minicpmo_4_5/results/*.json +# commands: +# - export SEED_TTS_WER_EVAL=1 +# - export SEED_TTS_EVAL_DEVICE=cuda:1 +# - pytest -s -v tests/e2e/accuracy/minicpmo_4_5/test_minicpmo_4_5.py -m "full_model and H100 and B200 and cuda and (cards_1 or cards_2)" --run-level full_model + +# - label: ":full_moon: Omni · Perf Test · No Async Chunk" +# source_file_dependencies: omni_qwen3_omni_perf +# key: nightly-omni-performance-no-async-chunk +# timeout_in_minutes: 300 +# artifact_paths: +# - tests/dfx/perf/results/*.json +# commands: +# - export BENCHMARK_DIR=tests/dfx/perf/results +# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_no_async_chunk.json -m "H100 and B200 and cards_2" + +# - label: ":full_moon: Omni · Perf Test · Async Chunk" +# source_file_dependencies: omni_qwen3_omni_perf +# key: nightly-omni-performance-async-chunk +# timeout_in_minutes: 180 +# artifact_paths: +# - tests/dfx/perf/results/*.json +# commands: +# - export BENCHMARK_DIR=tests/dfx/perf/results +# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_async_chunk.json -m "H100 and B200 and full_model and cards_2" + +# - label: ":full_moon: Omni · MiniCPM-o 4.5 · Perf Test" +# source_file_dependencies: omni_minicpmo_4_5_perf +# key: nightly-omni-performance-minicpmo-4-5 +# timeout_in_minutes: 180 +# artifact_paths: +# - tests/dfx/perf/results/*.json +# commands: +# - export BENCHMARK_DIR=tests/dfx/perf/results +# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minicpmo_4_5.json -m "H100 and B200 and cards_1" + +# - label: ":full_moon: Omni · MiniCPM-o 4.5 · Duplex Seed-TTS Perf Test" +# source_file_dependencies: omni_minicpmo_4_5_duplex_perf +# key: nightly-omni-performance-minicpmo-4-5-duplex-seed-tts +# timeout_in_minutes: 180 +# artifact_paths: +# - tests/dfx/perf/results/*.json +# commands: +# - export BENCHMARK_DIR=tests/dfx/perf/results +# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minicpmo_4_5_duplex_seed_tts.json -m "H100 and B200 and cards_1" + +# - label: ":full_moon: Omni · Multi-Replica Startup Test with 4x H100" +# source_file_dependencies: omni_qwen3_omni_function +# timeout_in_minutes: 45 +# commands: +# - pytest -s -v tests/e2e/online_serving/test_qwen3_omni_multi_replicas.py -m "full_model and H100 and B200 and cards_4" --run-level "core_model" - group: ":card_index_dividers: TTS Model Test" key: nightly-tts-test-group @@ -122,14 +123,14 @@ steps: if: build.env("NIGHTLY") == "1" || build.pull_request.labels includes "nightly-test" steps: - - label: ":full_moon: TTS · Function Test with L4" - source_file_dependencies: tts_qwen3_tts_function - timeout_in_minutes: 120 - commands: - - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" - - export VLLM_USE_DEEP_GEMM="0" - - export VLLM_MOE_USE_DEEP_GEMM="0" - - pytest -s -v tests/e2e/ -m "full_model and L4 and B200 and tts and cards_1" --run-level "full_model" --ignore=tests/e2e/accuracy +# - label: ":full_moon: TTS · Function Test with L4" +# source_file_dependencies: tts_qwen3_tts_function +# timeout_in_minutes: 120 +# commands: +# - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" +# - export VLLM_USE_DEEP_GEMM="0" +# - export VLLM_MOE_USE_DEEP_GEMM="0" +# - pytest -s -v tests/e2e/ -m "full_model and L4 and B200 and tts and cards_1" --run-level "full_model" --ignore=tests/e2e/accuracy - label: ":full_moon: TTS · Perf Test · Single-GPU" source_file_dependencies: tts_qwen3_tts_perf @@ -142,28 +143,28 @@ steps: - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_tts.json -m "H100 and B200 and cards_1" - - label: ":full_moon: TTS · Perf Test · 2-GPU" - source_file_dependencies: tts_qwen3_tts_perf - key: nightly-tts-performance-2gpu - timeout_in_minutes: 180 - artifact_paths: - - tests/dfx/perf/results/*.json - commands: - - export BENCHMARK_DIR=tests/dfx/perf/results - - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" - - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_tts.json -m "H100 and B200 and cards_2" - - - group: ":card_index_dividers: Tiny Model Tests [Multi-GPU]" - key: nightly-tiny-model-test-group - depends_on: upload-nightly-pipeline - if: >- - build.env("NIGHTLY") == "1" || - build.pull_request.labels includes "nightly-test" - steps: - - - label: ":full_moon: Tiny Model Tests · Diffusion (Multi-GPU accelerations) · 2-GPU" - source_file_dependencies: diffusion_tiny_model - timeout_in_minutes: 30 +# - label: ":full_moon: TTS · Perf Test · 2-GPU" +# source_file_dependencies: tts_qwen3_tts_perf +# key: nightly-tts-performance-2gpu +# timeout_in_minutes: 180 +# artifact_paths: +# - tests/dfx/perf/results/*.json +# commands: +# - export BENCHMARK_DIR=tests/dfx/perf/results +# - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" +# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_tts.json -m "H100 and B200 and cards_2" + +# - group: ":card_index_dividers: Tiny Model Tests [Multi-GPU]" +# key: nightly-tiny-model-test-group +# depends_on: upload-nightly-pipeline +# if: >- +# build.env("NIGHTLY") == "1" || +# build.pull_request.labels includes "nightly-test" +# steps: + +# - label: ":full_moon: Tiny Model Tests · Diffusion (Multi-GPU accelerations) · 2-GPU" +# source_file_dependencies: diffusion_tiny_model +# timeout_in_minutes: 30 # NOTE: Single-GPU (core_model) tiny model tests run on the ready label, # so nightly only adds the multi-GPU parallelism configurations. # @@ -171,357 +172,357 @@ steps: # --run-level `core_model` is used to indicate that we should use tiny model weights # for the CI, since these tests should pass irrespective of whether or not we use # the full weights or not. - commands: - - pytest -sv tests/model_tests/diffusion -m 'full_model and L4 and B200 and cuda and cards_2' --run-level "core_model" +# commands: +# - pytest -sv tests/model_tests/diffusion -m 'full_model and L4 and B200 and cuda and cards_2' --run-level "core_model" - - label: ":full_moon: Tiny Model Tests · Diffusion (Multi-GPU accelerations) · 4-GPU" - source_file_dependencies: diffusion_tiny_model - timeout_in_minutes: 30 +# - label: ":full_moon: Tiny Model Tests · Diffusion (Multi-GPU accelerations) · 4-GPU" +# source_file_dependencies: diffusion_tiny_model +# timeout_in_minutes: 30 # CFG+TP extra_test_groups request 2×2=4 devices (see get_required_device_count). - commands: - - pytest -sv tests/model_tests/diffusion -m 'full_model and L4 and B200 and cuda and cards_4' --run-level "core_model" - - - group: ":card_index_dividers: Diffusion X2I(&A&T) Model Test" - key: nightly-diffusion-x2iat-group - depends_on: upload-nightly-pipeline - if: >- - build.env("NIGHTLY") == "1" || - build.pull_request.labels includes "nightly-test" - steps: - - - label: ":full_moon: Diffusion X2I(&A&T) · Function Test with H100 · Single-GPU" - source_file_dependencies: diffusion_qwen_image_function - timeout_in_minutes: 120 - commands: - - >- - pytest -sv - tests/e2e/online_serving/test_qwen_image_expansion.py - -m "full_model and diffusion and H100 and B200 and cards_1" - --run-level "full_model" - - - label: ":robot_face: Robot Policy OpenPI · π0.5 · H100 · Single-GPU" - timeout_in_minutes: 120 - commands: - - pip install --no-deps "openpi-client @ git+https://github.com/Physical-Intelligence/openpi.git@215abfb217dbac7d5f1273282331b9b1866c0479#subdirectory=packages/openpi-client" - - >- - pytest -sv - tests/e2e/online_serving/test_pi05_expansion.py - -m "full_model and diffusion and H100 and cards_1" - --run-level "full_model" - mirror_hardwares: h100_1 - - - label: ":full_moon: Diffusion X2I(&A&T) · Function Test with H100 · 2-GPU" - source_file_dependencies: diffusion_qwen_image_function - timeout_in_minutes: 120 - commands: - - >- - pytest -sv - tests/e2e/online_serving/test_qwen_image_expansion.py - -m "full_model and diffusion and H100 and B200 and cards_2" - --run-level "full_model" - - - label: ":full_moon: Diffusion X2I(&A&T) · Function Test with L4 · Single-GPU" - source_file_dependencies: diffusion_qwen_image_function - timeout_in_minutes: 120 - commands: - - pytest -sv tests/e2e/offline_inference/test_qwen_image_autoround_w4a16_expansion.py -m "full_model and diffusion and L4 and B200 and cards_1" --run-level "full_model" - - - - label: ":full_moon: Diffusion X2I(&A&T) · Doc Test" - source_file_dependencies: diffusion_text_to_image_doc - timeout_in_minutes: 60 - commands: - - pytest -s -v tests/examples/*/test_text_to_image.py -m "full_model and example and H100 and B200 and (cards_1 or cards_2)" --run-level "full_model" - - - label: ":full_moon: Diffusion X2I(&A&T) · Accuracy Test" - source_file_dependencies: diffusion_qwen_image_accuracy - timeout_in_minutes: 180 - commands: - - export VLLM_HTTP_TIMEOUT_KEEP_ALIVE=120 - - pytest -s -v tests/e2e/accuracy/test_qwen_image.py -m "H100 and B200 and cards_1" --run-level full_model - - pytest -s -v tests/e2e/accuracy/test_diffusers_backend_similarity.py -k '2i_matches_diffusers' -m "full_model and H100 and B200 and cards_1" --run-level full_model - - - label: ":full_moon: HunyuanImage3-DIT · Accuracy Test" - source_file_dependencies: diffusion_hunyuan_image3_accuracy - timeout_in_minutes: 180 - commands: - - export HUNYUAN_IMAGE3_MODEL="tencent/HunyuanImage-3.0-Instruct" - - export HUNYUAN_IMAGE3_DEVICES="0,1,2,3" - - pytest -s -v tests/e2e/accuracy/test_hunyuan_image3_pixel_accuracy.py -m "full_model and H100 and B200 and cards_4" --run-level full_model - - - label: ":full_moon: Diffusion X2I(&A&T) · Perf Test · Qwen-Image · Single-GPU" - source_file_dependencies: diffusion_qwen_image_perf - key: nightly-diffusion-x2iat-performance-qwen-image-single - timeout_in_minutes: 180 - artifact_paths: - - tests/dfx/perf/results/diffusion_result_*.json - - tests/dfx/perf/results/logs/*.log - commands: - - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results - - export CACHE_DIT_VERSION=1.5.0 - - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_1" - - - label: ":full_moon: Diffusion X2I(&A&T) · Perf Test · Qwen-Image · 4-GPU" - source_file_dependencies: diffusion_qwen_image_perf - key: nightly-diffusion-x2iat-performance-qwen-image-4gpu - timeout_in_minutes: 180 - artifact_paths: - - tests/dfx/perf/results/diffusion_result_*.json - - tests/dfx/perf/results/logs/*.log - commands: - - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results - - export CACHE_DIT_VERSION=1.5.0 +# commands: +# - pytest -sv tests/model_tests/diffusion -m 'full_model and L4 and B200 and cuda and cards_4' --run-level "core_model" + +# - group: ":card_index_dividers: Diffusion X2I(&A&T) Model Test" +# key: nightly-diffusion-x2iat-group +# depends_on: upload-nightly-pipeline +# if: >- +# build.env("NIGHTLY") == "1" || +# build.pull_request.labels includes "nightly-test" +# steps: + +# - label: ":full_moon: Diffusion X2I(&A&T) · Function Test with H100 · Single-GPU" +# source_file_dependencies: diffusion_qwen_image_function +# timeout_in_minutes: 120 +# commands: +# - >- +# pytest -sv +# tests/e2e/online_serving/test_qwen_image_expansion.py +# -m "full_model and diffusion and H100 and B200 and cards_1" +# --run-level "full_model" + +# - label: ":robot_face: Robot Policy OpenPI · π0.5 · H100 · Single-GPU" +# timeout_in_minutes: 120 +# commands: +# - pip install --no-deps "openpi-client @ git+https://github.com/Physical-Intelligence/openpi.git@215abfb217dbac7d5f1273282331b9b1866c0479#subdirectory=packages/openpi-client" +# - >- +# pytest -sv +# tests/e2e/online_serving/test_pi05_expansion.py +# -m "full_model and diffusion and H100 and cards_1" +# --run-level "full_model" +# mirror_hardwares: h100_1 + +# - label: ":full_moon: Diffusion X2I(&A&T) · Function Test with H100 · 2-GPU" +# source_file_dependencies: diffusion_qwen_image_function +# timeout_in_minutes: 120 +# commands: +# - >- +# pytest -sv +# tests/e2e/online_serving/test_qwen_image_expansion.py +# -m "full_model and diffusion and H100 and B200 and cards_2" +# --run-level "full_model" + +# - label: ":full_moon: Diffusion X2I(&A&T) · Function Test with L4 · Single-GPU" +# source_file_dependencies: diffusion_qwen_image_function +# timeout_in_minutes: 120 +# commands: +# - pytest -sv tests/e2e/offline_inference/test_qwen_image_autoround_w4a16_expansion.py -m "full_model and diffusion and L4 and B200 and cards_1" --run-level "full_model" + + +# - label: ":full_moon: Diffusion X2I(&A&T) · Doc Test" +# source_file_dependencies: diffusion_text_to_image_doc +# timeout_in_minutes: 60 +# commands: +# - pytest -s -v tests/examples/*/test_text_to_image.py -m "full_model and example and H100 and B200 and (cards_1 or cards_2)" --run-level "full_model" + +# - label: ":full_moon: Diffusion X2I(&A&T) · Accuracy Test" +# source_file_dependencies: diffusion_qwen_image_accuracy +# timeout_in_minutes: 180 +# commands: +# - export VLLM_HTTP_TIMEOUT_KEEP_ALIVE=120 +# - pytest -s -v tests/e2e/accuracy/test_qwen_image.py -m "H100 and B200 and cards_1" --run-level full_model +# - pytest -s -v tests/e2e/accuracy/test_diffusers_backend_similarity.py -k '2i_matches_diffusers' -m "full_model and H100 and B200 and cards_1" --run-level full_model + +# - label: ":full_moon: HunyuanImage3-DIT · Accuracy Test" +# source_file_dependencies: diffusion_hunyuan_image3_accuracy +# timeout_in_minutes: 180 +# commands: +# - export HUNYUAN_IMAGE3_MODEL="tencent/HunyuanImage-3.0-Instruct" +# - export HUNYUAN_IMAGE3_DEVICES="0,1,2,3" +# - pytest -s -v tests/e2e/accuracy/test_hunyuan_image3_pixel_accuracy.py -m "full_model and H100 and B200 and cards_4" --run-level full_model + +# - label: ":full_moon: Diffusion X2I(&A&T) · Perf Test · Qwen-Image · Single-GPU" +# source_file_dependencies: diffusion_qwen_image_perf +# key: nightly-diffusion-x2iat-performance-qwen-image-single +# timeout_in_minutes: 180 +# artifact_paths: +# - tests/dfx/perf/results/diffusion_result_*.json +# - tests/dfx/perf/results/logs/*.log +# commands: +# - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results +# - export CACHE_DIT_VERSION=1.5.0 +# - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_1" + +# - label: ":full_moon: Diffusion X2I(&A&T) · Perf Test · Qwen-Image · 4-GPU" +# source_file_dependencies: diffusion_qwen_image_perf +# key: nightly-diffusion-x2iat-performance-qwen-image-4gpu +# timeout_in_minutes: 180 +# artifact_paths: +# - tests/dfx/perf/results/diffusion_result_*.json +# - tests/dfx/perf/results/logs/*.log +# commands: +# - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results +# - export CACHE_DIT_VERSION=1.5.0 # Do not pin DIFFUSION_ATTENTION_BACKEND: H100 auto-selects FLASH_ATTN; # B200 rejects explicit FLASH_ATTN without FA4 and uses CUDNN/TRTLLM. - - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_4" - - - group: ":card_index_dividers: Diffusion X2V Model Test" - key: nightly-diffusion-x2v-group - depends_on: upload-nightly-pipeline - if: >- - build.env("NIGHTLY") == "1" || - build.pull_request.labels includes "nightly-test" - steps: - - - label: ":full_moon: Diffusion X2V · Wan2.2 T2V Function Test · Single-GPU" - source_file_dependencies: diffusion_wan22_function - timeout_in_minutes: 90 - commands: - - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "t2v" -m "full_model and cuda and H100 and B200 and cards_1" --run-level "full_model" - - - label: ":full_moon: Diffusion X2V · Wan2.2 T2V Function Test · 2-GPU" - source_file_dependencies: diffusion_wan22_function - timeout_in_minutes: 90 - commands: - - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "t2v" -m "full_model and cuda and H100 and B200 and cards_2" --run-level "full_model" - - - label: ":full_moon: Diffusion X2V · Wan2.2 I2V/TI2V Function Test · Single-GPU" - source_file_dependencies: diffusion_wan22_function - timeout_in_minutes: 90 - commands: - - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "i2v" -m "full_model and cuda and H100 and B200 and cards_1" --run-level "full_model" - - - label: ":full_moon: Diffusion X2V · Wan2.2 I2V/TI2V Function Test · 2-GPU" - source_file_dependencies: diffusion_wan22_function - timeout_in_minutes: 90 - commands: - - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "i2v" -m "full_model and cuda and H100 and B200 and cards_2" --run-level "full_model" - - - label: ":full_moon: Diffusion X2V · HunyuanVideo-1.5 Function Test" - source_file_dependencies: diffusion_hunyuan_video_function - timeout_in_minutes: 90 - commands: - - pytest -s -v tests/e2e/online_serving/test_hunyuan_video_15_expansion.py -m "full_model and H100 and B200 and cuda and (cards_1 or cards_2)" --run-level "full_model" - - - label: ":full_moon: Diffusion X2V · LingBot Function Test" - source_file_dependencies: diffusion_lingbot_function - timeout_in_minutes: 120 - commands: - - >- - pytest -s -v - tests/e2e/online_serving/test_lingbot_video.py - tests/e2e/online_serving/test_lingbot_video_moe.py - tests/e2e/offline_inference/test_lingbot_world_v2.py - -m "full_model and diffusion and H100 and B200 and cards_1" - --run-level "full_model" - - - label: ":full_moon: Diffusion X2V · Doc Test" - source_file_dependencies: diffusion_image_to_video_doc - timeout_in_minutes: 60 - commands: - - pytest -s -v tests/examples/offline_inference/test_image_to_video.py -m "full_model and example and H100 and B200 and cards_1" --run-level "full_model" - - - label: ":full_moon: Diffusion X2V · Wan2.2 Autoround Function Test" - source_file_dependencies: diffusion_wan22_function - timeout_in_minutes: 90 - commands: - - pytest -s -v tests/e2e/offline_inference/test_wan22_autoround_w4a16_expansion.py -m "full_model and H100 and B200 and cuda and (cards_1 or cards_2)" --run-level "full_model" - - - label: ":full_moon: Diffusion X2V · Wan2.2 Device-Postprocess Equivalence Test" - source_file_dependencies: diffusion_wan22_function - timeout_in_minutes: 90 - commands: - - | - VLLM_OMNI_DEVICE_POSTPROCESS_MODEL=Wan-AI/Wan2.2-TI2V-5B-Diffusers \ - pytest -s -v tests/e2e/offline_inference/test_device_postprocess_equivalence.py -m "full_model and H100 and B200 and cuda and cards_1" --run-level "full_model" - - - label: ":full_moon: Diffusion X2V · Wan2.2 Accuracy Test" - source_file_dependencies: diffusion_wan22_accuracy - timeout_in_minutes: 180 - commands: - - pytest -s -v tests/e2e/accuracy/wan22_i2v/test_wan22_i2v_video_similarity.py -m "full_model and H100 and B200 and (cards_1 or cards_2)" --run-level full_model - - pytest -s -v tests/e2e/accuracy/test_diffusers_backend_similarity.py -k '2v_matches_diffusers' -m "full_model and H100 and B200 and cards_1" --run-level full_model - - - label: ":full_moon: Diffusion X2V · MiniMax H3 I2VA/Ref2VA Accuracy Test" - source_file_dependencies: diffusion_minimax_h3_accuracy - timeout_in_minutes: 180 - commands: - - pytest -s -v tests/e2e/accuracy/minimax_h3/test_minimax_h3_i2va_ref2va_similarity.py -m "full_model and H100 and B200 and cards_4" --run-level full_model - - - label: ":full_moon: Diffusion X2V · HunyuanVideo-1.5 Accuracy Test" - source_file_dependencies: diffusion_hunyuan_video_accuracy - timeout_in_minutes: 180 - commands: - - pytest -s -v tests/e2e/accuracy/hunyuanvideo15_t2v/test_hunyuanvideo15_t2v_video_similarity.py -m "full_model and H100 and B200 and (cards_1 or cards_2)" --run-level full_model - - pytest -s -v tests/e2e/accuracy/hunyuanvideo15_i2v/test_hunyuanvideo15_i2v_video_similarity.py -m "full_model and H100 and B200 and (cards_1 or cards_2)" --run-level full_model - - - label: ":full_moon: Diffusion X2V · Perf Test · Single-GPU" - source_file_dependencies: diffusion_wan22_perf - key: nightly-diffusion-x2v-performance-single - timeout_in_minutes: 180 - artifact_paths: - - tests/dfx/perf/results/*.json - commands: - - export BENCHMARK_DIR=tests/dfx/perf/results - - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json -m "H100 and B200 and cards_1" - - - label: ":full_moon: Diffusion X2V · Perf Test · 2-GPU" - source_file_dependencies: - - diffusion_wan22_perf - - diffusion_cosmos3_perf - key: nightly-diffusion-x2v-performance-2gpu - timeout_in_minutes: 180 - artifact_paths: - - tests/dfx/perf/results/*.json - commands: - - export BENCHMARK_DIR=tests/dfx/perf/results - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json -m "H100 and B200 and cards_2" - EXIT1=$$? - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_cosmos3_vllm_omni.json -m "H100 and B200 and cards_2" - EXIT2=$$? - if [ $$EXIT1 -ne 0 ]; then exit $$EXIT1; fi - exit $$EXIT2 - - - label: ":full_moon: Diffusion X2V · Perf Test · HunyuanVideo-1.5" - source_file_dependencies: diffusion_hunyuan_video_perf - key: nightly-diffusion-x2v-performance-hunyuanvideo15 - timeout_in_minutes: 180 - artifact_paths: - - tests/dfx/perf/results/*.json - commands: - - export BENCHMARK_DIR=tests/dfx/perf/results - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_hunyuanvideo15_t2v_vllm_omni.json -m "H100 and B200 and (cards_1 or cards_2)" - EXIT1=$$? - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_hunyuanvideo15_i2v_vllm_omni.json -m "H100 and B200 and (cards_1 or cards_2)" - EXIT2=$$? - if [ $$EXIT1 -ne 0 ]; then exit $$EXIT1; fi - exit $$EXIT2 - - - label: ":full_moon: Diffusion X2V · Perf Test · MiniMax-H3" - source_file_dependencies: diffusion_minimax_h3_perf - key: nightly-diffusion-x2v-performance-minimax-h3 - timeout_in_minutes: 180 - artifact_paths: - - tests/dfx/perf/results/*.json - commands: - - export BENCHMARK_DIR=tests/dfx/perf/results - - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json -m "H100 and B200 and cards_4" - - - label: ":full_moon: Diffusion X2V · Perf Test · LingBot" - source_file_dependencies: diffusion_lingbot_perf - key: nightly-diffusion-x2v-performance-lingbot - timeout_in_minutes: 180 - artifact_paths: - - tests/dfx/perf/results/*.json - commands: - - export BENCHMARK_DIR=tests/dfx/perf/results - - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json -m "H100 and B200 and cards_1" - - - group: ":card_index_dividers: Diffusion Test" - key: nightly-diffusion-test-group - depends_on: upload-nightly-pipeline - if: build.env("NIGHTLY") == "1" || build.pull_request.labels includes "nightly-test" - steps: - - label: ":full_moon: Diffusion · Distributed Test with L4 · 2-GPU" - source_file_dependencies: diffusion_distributed_attention - timeout_in_minutes: 60 - commands: - - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and L4 and B200 and cards_2' --run-level "full_model" - - - label: ":full_moon: Diffusion · Distributed Test with L4 · 3-GPU" - source_file_dependencies: diffusion_distributed_attention - timeout_in_minutes: 60 - commands: - - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and L4 and B200 and cards_3' --run-level "full_model" - - - label: ":full_moon: Diffusion · Distributed Test with L4 · 4-GPU" - source_file_dependencies: diffusion_distributed_attention - timeout_in_minutes: 60 - commands: - - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and L4 and B200 and cards_4' --run-level "full_model" - - - label: ":full_moon: Diffusion · Distributed Test with H100 · 2-GPU" - source_file_dependencies: diffusion_distributed_attention - timeout_in_minutes: 60 - commands: - - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and H100 and B200 and cards_2' --run-level "full_model" - - - label: ":full_moon: Diffusion · Distributed Test with H100 · 4-GPU" - source_file_dependencies: diffusion_distributed_attention - timeout_in_minutes: 60 - commands: - - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and H100 and B200 and cards_4' --run-level "full_model" - - - label: ":full_moon: Diffusion · Offloader Test" - source_file_dependencies: diffusion_offloader - timeout_in_minutes: 60 - commands: - - pytest -sv tests/diffusion/offloader -m "full_model and L4 and B200 and cuda and cards_1" - - - label: ":full_moon: Diffusion · Quantization Test with H100 · Single-GPU" - source_file_dependencies: diffusion_quantization - timeout_in_minutes: 60 - commands: - - pytest -sv tests/diffusion/quantization -m 'full_model and cuda and H100 and B200 and cards_1' --run-level "full_model" - - - label: ":full_moon: Diffusion · MiniMax H3 FP8 Quality Test with H100 · 2-GPU" - source_file_dependencies: diffusion_minimax_h3_function - timeout_in_minutes: 60 - commands: - - pytest -sv tests/diffusion/models/minimax_h3/test_minimax_h3_quantization_quality.py -m 'full_model and cuda and H100 and B200 and cards_2' --run-level "full_model" - - - label: ":full_moon: Diffusion · Quantization Test with L4" - source_file_dependencies: diffusion_quantization - timeout_in_minutes: 60 - commands: - - pip install "vllm-gguf-plugin==0.0.4" - - pytest -sv tests/diffusion/quantization -m 'full_model and cuda and L4 and B200 and cards_1' --run-level "full_model" +# - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_4" + +# - group: ":card_index_dividers: Diffusion X2V Model Test" +# key: nightly-diffusion-x2v-group +# depends_on: upload-nightly-pipeline +# if: >- +# build.env("NIGHTLY") == "1" || +# build.pull_request.labels includes "nightly-test" +# steps: + +# - label: ":full_moon: Diffusion X2V · Wan2.2 T2V Function Test · Single-GPU" +# source_file_dependencies: diffusion_wan22_function +# timeout_in_minutes: 90 +# commands: +# - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "t2v" -m "full_model and cuda and H100 and B200 and cards_1" --run-level "full_model" + +# - label: ":full_moon: Diffusion X2V · Wan2.2 T2V Function Test · 2-GPU" +# source_file_dependencies: diffusion_wan22_function +# timeout_in_minutes: 90 +# commands: +# - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "t2v" -m "full_model and cuda and H100 and B200 and cards_2" --run-level "full_model" + +# - label: ":full_moon: Diffusion X2V · Wan2.2 I2V/TI2V Function Test · Single-GPU" +# source_file_dependencies: diffusion_wan22_function +# timeout_in_minutes: 90 +# commands: +# - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "i2v" -m "full_model and cuda and H100 and B200 and cards_1" --run-level "full_model" + +# - label: ":full_moon: Diffusion X2V · Wan2.2 I2V/TI2V Function Test · 2-GPU" +# source_file_dependencies: diffusion_wan22_function +# timeout_in_minutes: 90 +# commands: +# - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "i2v" -m "full_model and cuda and H100 and B200 and cards_2" --run-level "full_model" + +# - label: ":full_moon: Diffusion X2V · HunyuanVideo-1.5 Function Test" +# source_file_dependencies: diffusion_hunyuan_video_function +# timeout_in_minutes: 90 +# commands: +# - pytest -s -v tests/e2e/online_serving/test_hunyuan_video_15_expansion.py -m "full_model and H100 and B200 and cuda and (cards_1 or cards_2)" --run-level "full_model" + +# - label: ":full_moon: Diffusion X2V · LingBot Function Test" +# source_file_dependencies: diffusion_lingbot_function +# timeout_in_minutes: 120 +# commands: +# - >- +# pytest -s -v +# tests/e2e/online_serving/test_lingbot_video.py +# tests/e2e/online_serving/test_lingbot_video_moe.py +# tests/e2e/offline_inference/test_lingbot_world_v2.py +# -m "full_model and diffusion and H100 and B200 and cards_1" +# --run-level "full_model" + +# - label: ":full_moon: Diffusion X2V · Doc Test" +# source_file_dependencies: diffusion_image_to_video_doc +# timeout_in_minutes: 60 +# commands: +# - pytest -s -v tests/examples/offline_inference/test_image_to_video.py -m "full_model and example and H100 and B200 and cards_1" --run-level "full_model" + +# - label: ":full_moon: Diffusion X2V · Wan2.2 Autoround Function Test" +# source_file_dependencies: diffusion_wan22_function +# timeout_in_minutes: 90 +# commands: +# - pytest -s -v tests/e2e/offline_inference/test_wan22_autoround_w4a16_expansion.py -m "full_model and H100 and B200 and cuda and (cards_1 or cards_2)" --run-level "full_model" + +# - label: ":full_moon: Diffusion X2V · Wan2.2 Device-Postprocess Equivalence Test" +# source_file_dependencies: diffusion_wan22_function +# timeout_in_minutes: 90 +# commands: +# - | +# VLLM_OMNI_DEVICE_POSTPROCESS_MODEL=Wan-AI/Wan2.2-TI2V-5B-Diffusers \ +# pytest -s -v tests/e2e/offline_inference/test_device_postprocess_equivalence.py -m "full_model and H100 and B200 and cuda and cards_1" --run-level "full_model" + +# - label: ":full_moon: Diffusion X2V · Wan2.2 Accuracy Test" +# source_file_dependencies: diffusion_wan22_accuracy +# timeout_in_minutes: 180 +# commands: +# - pytest -s -v tests/e2e/accuracy/wan22_i2v/test_wan22_i2v_video_similarity.py -m "full_model and H100 and B200 and (cards_1 or cards_2)" --run-level full_model +# - pytest -s -v tests/e2e/accuracy/test_diffusers_backend_similarity.py -k '2v_matches_diffusers' -m "full_model and H100 and B200 and cards_1" --run-level full_model + +# - label: ":full_moon: Diffusion X2V · MiniMax H3 I2VA/Ref2VA Accuracy Test" +# source_file_dependencies: diffusion_minimax_h3_accuracy +# timeout_in_minutes: 180 +# commands: +# - pytest -s -v tests/e2e/accuracy/minimax_h3/test_minimax_h3_i2va_ref2va_similarity.py -m "full_model and H100 and B200 and cards_4" --run-level full_model + +# - label: ":full_moon: Diffusion X2V · HunyuanVideo-1.5 Accuracy Test" +# source_file_dependencies: diffusion_hunyuan_video_accuracy +# timeout_in_minutes: 180 +# commands: +# - pytest -s -v tests/e2e/accuracy/hunyuanvideo15_t2v/test_hunyuanvideo15_t2v_video_similarity.py -m "full_model and H100 and B200 and (cards_1 or cards_2)" --run-level full_model +# - pytest -s -v tests/e2e/accuracy/hunyuanvideo15_i2v/test_hunyuanvideo15_i2v_video_similarity.py -m "full_model and H100 and B200 and (cards_1 or cards_2)" --run-level full_model + +# - label: ":full_moon: Diffusion X2V · Perf Test · Single-GPU" +# source_file_dependencies: diffusion_wan22_perf +# key: nightly-diffusion-x2v-performance-single +# timeout_in_minutes: 180 +# artifact_paths: +# - tests/dfx/perf/results/*.json +# commands: +# - export BENCHMARK_DIR=tests/dfx/perf/results +# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json -m "H100 and B200 and cards_1" + +# - label: ":full_moon: Diffusion X2V · Perf Test · 2-GPU" +# source_file_dependencies: +# - diffusion_wan22_perf +# - diffusion_cosmos3_perf +# key: nightly-diffusion-x2v-performance-2gpu +# timeout_in_minutes: 180 +# artifact_paths: +# - tests/dfx/perf/results/*.json +# commands: +# - export BENCHMARK_DIR=tests/dfx/perf/results +# - | +# set +e +# pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json -m "H100 and B200 and cards_2" +# EXIT1=$$? +# pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_cosmos3_vllm_omni.json -m "H100 and B200 and cards_2" +# EXIT2=$$? +# if [ $$EXIT1 -ne 0 ]; then exit $$EXIT1; fi +# exit $$EXIT2 + +# - label: ":full_moon: Diffusion X2V · Perf Test · HunyuanVideo-1.5" +# source_file_dependencies: diffusion_hunyuan_video_perf +# key: nightly-diffusion-x2v-performance-hunyuanvideo15 +# timeout_in_minutes: 180 +# artifact_paths: +# - tests/dfx/perf/results/*.json +# commands: +# - export BENCHMARK_DIR=tests/dfx/perf/results +# - | +# set +e +# pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_hunyuanvideo15_t2v_vllm_omni.json -m "H100 and B200 and (cards_1 or cards_2)" +# EXIT1=$$? +# pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_hunyuanvideo15_i2v_vllm_omni.json -m "H100 and B200 and (cards_1 or cards_2)" +# EXIT2=$$? +# if [ $$EXIT1 -ne 0 ]; then exit $$EXIT1; fi +# exit $$EXIT2 + +# - label: ":full_moon: Diffusion X2V · Perf Test · MiniMax-H3" +# source_file_dependencies: diffusion_minimax_h3_perf +# key: nightly-diffusion-x2v-performance-minimax-h3 +# timeout_in_minutes: 180 +# artifact_paths: +# - tests/dfx/perf/results/*.json +# commands: +# - export BENCHMARK_DIR=tests/dfx/perf/results +# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json -m "H100 and B200 and cards_4" + +# - label: ":full_moon: Diffusion X2V · Perf Test · LingBot" +# source_file_dependencies: diffusion_lingbot_perf +# key: nightly-diffusion-x2v-performance-lingbot +# timeout_in_minutes: 180 +# artifact_paths: +# - tests/dfx/perf/results/*.json +# commands: +# - export BENCHMARK_DIR=tests/dfx/perf/results +# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json -m "H100 and B200 and cards_1" + +# - group: ":card_index_dividers: Diffusion Test" +# key: nightly-diffusion-test-group +# depends_on: upload-nightly-pipeline +# if: build.env("NIGHTLY") == "1" || build.pull_request.labels includes "nightly-test" +# steps: +# - label: ":full_moon: Diffusion · Distributed Test with L4 · 2-GPU" +# source_file_dependencies: diffusion_distributed_attention +# timeout_in_minutes: 60 +# commands: +# - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and L4 and B200 and cards_2' --run-level "full_model" + +# - label: ":full_moon: Diffusion · Distributed Test with L4 · 3-GPU" +# source_file_dependencies: diffusion_distributed_attention +# timeout_in_minutes: 60 +# commands: +# - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and L4 and B200 and cards_3' --run-level "full_model" + +# - label: ":full_moon: Diffusion · Distributed Test with L4 · 4-GPU" +# source_file_dependencies: diffusion_distributed_attention +# timeout_in_minutes: 60 +# commands: +# - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and L4 and B200 and cards_4' --run-level "full_model" + +# - label: ":full_moon: Diffusion · Distributed Test with H100 · 2-GPU" +# source_file_dependencies: diffusion_distributed_attention +# timeout_in_minutes: 60 +# commands: +# - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and H100 and B200 and cards_2' --run-level "full_model" + +# - label: ":full_moon: Diffusion · Distributed Test with H100 · 4-GPU" +# source_file_dependencies: diffusion_distributed_attention +# timeout_in_minutes: 60 +# commands: +# - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and H100 and B200 and cards_4' --run-level "full_model" + +# - label: ":full_moon: Diffusion · Offloader Test" +# source_file_dependencies: diffusion_offloader +# timeout_in_minutes: 60 +# commands: +# - pytest -sv tests/diffusion/offloader -m "full_model and L4 and B200 and cuda and cards_1" + +# - label: ":full_moon: Diffusion · Quantization Test with H100 · Single-GPU" +# source_file_dependencies: diffusion_quantization +# timeout_in_minutes: 60 +# commands: +# - pytest -sv tests/diffusion/quantization -m 'full_model and cuda and H100 and B200 and cards_1' --run-level "full_model" + +# - label: ":full_moon: Diffusion · MiniMax H3 FP8 Quality Test with H100 · 2-GPU" +# source_file_dependencies: diffusion_minimax_h3_function +# timeout_in_minutes: 60 +# commands: +# - pytest -sv tests/diffusion/models/minimax_h3/test_minimax_h3_quantization_quality.py -m 'full_model and cuda and H100 and B200 and cards_2' --run-level "full_model" + +# - label: ":full_moon: Diffusion · Quantization Test with L4" +# source_file_dependencies: diffusion_quantization +# timeout_in_minutes: 60 +# commands: +# - pip install "vllm-gguf-plugin==0.0.4" +# - pytest -sv tests/diffusion/quantization -m 'full_model and cuda and L4 and B200 and cards_1' --run-level "full_model" # NOTE: ":email: Nightly Collection & Email" is deprecated and will be removed soon. # Do not add new depends_on entries or extend this step. - - label: ":email: Nightly Collection & Email" - key: nightly-perf-distribution - depends_on: - - nightly-omni-performance-no-async-chunk - - nightly-omni-performance-async-chunk - - nightly-tts-performance-single - - nightly-tts-performance-2gpu - - nightly-diffusion-x2iat-performance-qwen-image-single - - nightly-diffusion-x2iat-performance-qwen-image-4gpu - - nightly-diffusion-x2v-performance-single - - nightly-diffusion-x2v-performance-2gpu - if: build.env("NIGHTLY") == "1" && build.env("EMAIL_DISTRIBUTION") == "1" - artifact_paths: - - tests/dfx/perf/results/*.xlsx - - tests/dfx/perf/results/*.html - commands: - - pip install openpyxl - - export DEFAULT_INPUT_DIR=tests/dfx/perf/results - - export DEFAULT_OUTPUT_DIR=tests/dfx/perf/results - - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-tts-performance-single - - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-tts-performance-2gpu - - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-omni-performance-no-async-chunk - - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-omni-performance-async-chunk - - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2iat-performance-qwen-image-single - - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2iat-performance-qwen-image-4gpu - - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2v-performance-single - - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2v-performance-2gpu - - python tools/nightly/generate_nightly_perf_excel.py - - python tools/nightly/generate_nightly_perf_html.py - - python tools/nightly/send_nightly_email.py --report-file "tests/dfx/perf/results/*.xlsx, tests/dfx/perf/results/*.html" - agents: - queue: "cpu_queue_premerge" +# - label: ":email: Nightly Collection & Email" +# key: nightly-perf-distribution +# depends_on: +# - nightly-omni-performance-no-async-chunk +# - nightly-omni-performance-async-chunk +# - nightly-tts-performance-single +# - nightly-tts-performance-2gpu +# - nightly-diffusion-x2iat-performance-qwen-image-single +# - nightly-diffusion-x2iat-performance-qwen-image-4gpu +# - nightly-diffusion-x2v-performance-single +# - nightly-diffusion-x2v-performance-2gpu +# if: build.env("NIGHTLY") == "1" && build.env("EMAIL_DISTRIBUTION") == "1" +# artifact_paths: +# - tests/dfx/perf/results/*.xlsx +# - tests/dfx/perf/results/*.html +# commands: +# - pip install openpyxl +# - export DEFAULT_INPUT_DIR=tests/dfx/perf/results +# - export DEFAULT_OUTPUT_DIR=tests/dfx/perf/results +# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-tts-performance-single +# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-tts-performance-2gpu +# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-omni-performance-no-async-chunk +# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-omni-performance-async-chunk +# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2iat-performance-qwen-image-single +# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2iat-performance-qwen-image-4gpu +# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2v-performance-single +# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2v-performance-2gpu +# - python tools/nightly/generate_nightly_perf_excel.py +# - python tools/nightly/generate_nightly_perf_html.py +# - python tools/nightly/send_nightly_email.py --report-file "tests/dfx/perf/results/*.xlsx, tests/dfx/perf/results/*.html" +# agents: +# queue: "cpu_queue_premerge" From 1d818607b47471f9b38d0d6f3fd2116b29b402a2 Mon Sep 17 00:00:00 2001 From: wangyu <410167048@qq.com> Date: Sun, 20 Sep 2026 14:14:55 +0800 Subject: [PATCH 12/15] remove debug Signed-off-by: wangyu <410167048@qq.com> --- .buildkite/cuda/test-nightly.yml | 967 +++++++++++++++---------------- 1 file changed, 483 insertions(+), 484 deletions(-) diff --git a/.buildkite/cuda/test-nightly.yml b/.buildkite/cuda/test-nightly.yml index c178dadb79d..a46c8f8bb86 100644 --- a/.buildkite/cuda/test-nightly.yml +++ b/.buildkite/cuda/test-nightly.yml @@ -8,114 +8,113 @@ env: HF_HUB_ETAG_TIMEOUT: 60 steps: - # TEMPORARY: only TTS Perf Test Single-GPU is active; all other jobs commented out. # Group: collapses under one heading in the Buildkite UI; child steps still run in parallel. -# - group: ":card_index_dividers: Omni Model Test" -# key: nightly-omni-test-group -# depends_on: upload-nightly-pipeline -# if: build.env("NIGHTLY") == "1" || build.pull_request.labels includes "nightly-test" -# steps: - -# - label: ":full_moon: Omni · Function Test with H100 · Single-GPU" -# source_file_dependencies: -# - omni_qwen3_omni_function -# - omni_minicpmo_4_5_function -# timeout_in_minutes: 120 -# commands: -# - pytest -sv tests/e2e --run-level "full_model" -m "full_model and H100 and B200 and omni and cards_1" --ignore=tests/e2e/accuracy --ignore=tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py - -# - label: ":full_moon: Omni · Function Test with H100 · 2-GPU" -# source_file_dependencies: -# - omni_qwen3_omni_function -# - omni_minicpmo_4_5_function -# timeout_in_minutes: 90 -# commands: -# - pytest -sv tests/e2e --run-level "full_model" -m "full_model and H100 and B200 and omni and cards_2" --ignore=tests/e2e/accuracy --ignore=tests/e2e/online_serving/test_qwen3_omni_multi_replicas.py - -# - label: ":full_moon: Omni · MiniCPM-o 4.5 Duplex Test" -# source_file_dependencies: omni_minicpmo_4_5_duplex_function -# timeout_in_minutes: 50 -# commands: -# - pytest -s -v tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py -m "full_model and cuda and H100 and B200 and omni and cards_1" --run-level "full_model" - -# - label: ":full_moon: Omni · Doc Test with L4 · 4-GPU" -# source_file_dependencies: omni_qwen2_5_omni_doc -# timeout_in_minutes: 90 -# commands: -# - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" -# - pytest -s -v tests/examples/ -m "full_model and omni and L4 and B200 and cards_4" --run-level "full_model" - -# - label: ":full_moon: Omni · Doc Test with H100 · 2-GPU" -# source_file_dependencies: omni_qwen3_omni_doc -# timeout_in_minutes: 90 -# commands: -# - pytest -s -v tests/examples/ -m "full_model and omni and H100 and B200 and cards_2" --run-level "full_model" - -# - label: ":full_moon: Omni · Accuracy Test" -# source_file_dependencies: omni_qwen3_omni_accuracy -# timeout_in_minutes: 180 -# artifact_paths: -# - tests/e2e/accuracy/qwen3_omni/results/qwen_omni_acc/*.json -# commands: -# - export SEED_TTS_WER_EVAL=1 -# - export SEED_TTS_EVAL_DEVICE=cuda:1 -# - pytest -s -v tests/e2e/accuracy/qwen3_omni/test_qwen3_omni.py -m "full_model and H100 and B200 and cards_2" --run-level full_model - -# - label: ":full_moon: Omni · MiniCPM-o 4.5 · Accuracy Test" -# source_file_dependencies: omni_minicpmo_4_5_accuracy -# timeout_in_minutes: 180 -# artifact_paths: -# - tests/e2e/accuracy/minicpmo_4_5/results/*.json -# commands: -# - export SEED_TTS_WER_EVAL=1 -# - export SEED_TTS_EVAL_DEVICE=cuda:1 -# - pytest -s -v tests/e2e/accuracy/minicpmo_4_5/test_minicpmo_4_5.py -m "full_model and H100 and B200 and cuda and (cards_1 or cards_2)" --run-level full_model - -# - label: ":full_moon: Omni · Perf Test · No Async Chunk" -# source_file_dependencies: omni_qwen3_omni_perf -# key: nightly-omni-performance-no-async-chunk -# timeout_in_minutes: 300 -# artifact_paths: -# - tests/dfx/perf/results/*.json -# commands: -# - export BENCHMARK_DIR=tests/dfx/perf/results -# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_no_async_chunk.json -m "H100 and B200 and cards_2" - -# - label: ":full_moon: Omni · Perf Test · Async Chunk" -# source_file_dependencies: omni_qwen3_omni_perf -# key: nightly-omni-performance-async-chunk -# timeout_in_minutes: 180 -# artifact_paths: -# - tests/dfx/perf/results/*.json -# commands: -# - export BENCHMARK_DIR=tests/dfx/perf/results -# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_async_chunk.json -m "H100 and B200 and full_model and cards_2" - -# - label: ":full_moon: Omni · MiniCPM-o 4.5 · Perf Test" -# source_file_dependencies: omni_minicpmo_4_5_perf -# key: nightly-omni-performance-minicpmo-4-5 -# timeout_in_minutes: 180 -# artifact_paths: -# - tests/dfx/perf/results/*.json -# commands: -# - export BENCHMARK_DIR=tests/dfx/perf/results -# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minicpmo_4_5.json -m "H100 and B200 and cards_1" - -# - label: ":full_moon: Omni · MiniCPM-o 4.5 · Duplex Seed-TTS Perf Test" -# source_file_dependencies: omni_minicpmo_4_5_duplex_perf -# key: nightly-omni-performance-minicpmo-4-5-duplex-seed-tts -# timeout_in_minutes: 180 -# artifact_paths: -# - tests/dfx/perf/results/*.json -# commands: -# - export BENCHMARK_DIR=tests/dfx/perf/results -# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minicpmo_4_5_duplex_seed_tts.json -m "H100 and B200 and cards_1" - -# - label: ":full_moon: Omni · Multi-Replica Startup Test with 4x H100" -# source_file_dependencies: omni_qwen3_omni_function -# timeout_in_minutes: 45 -# commands: -# - pytest -s -v tests/e2e/online_serving/test_qwen3_omni_multi_replicas.py -m "full_model and H100 and B200 and cards_4" --run-level "core_model" + - group: ":card_index_dividers: Omni Model Test" + key: nightly-omni-test-group + depends_on: upload-nightly-pipeline + if: build.env("NIGHTLY") == "1" || build.pull_request.labels includes "nightly-test" + steps: + + - label: ":full_moon: Omni · Function Test with H100 · Single-GPU" + source_file_dependencies: + - omni_qwen3_omni_function + - omni_minicpmo_4_5_function + timeout_in_minutes: 120 + commands: + - pytest -sv tests/e2e --run-level "full_model" -m "full_model and H100 and B200 and omni and cards_1" --ignore=tests/e2e/accuracy --ignore=tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py + + - label: ":full_moon: Omni · Function Test with H100 · 2-GPU" + source_file_dependencies: + - omni_qwen3_omni_function + - omni_minicpmo_4_5_function + timeout_in_minutes: 90 + commands: + - pytest -sv tests/e2e --run-level "full_model" -m "full_model and H100 and B200 and omni and cards_2" --ignore=tests/e2e/accuracy --ignore=tests/e2e/online_serving/test_qwen3_omni_multi_replicas.py + + - label: ":full_moon: Omni · MiniCPM-o 4.5 Duplex Test" + source_file_dependencies: omni_minicpmo_4_5_duplex_function + timeout_in_minutes: 50 + commands: + - pytest -s -v tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py -m "full_model and cuda and H100 and B200 and omni and cards_1" --run-level "full_model" + + - label: ":full_moon: Omni · Doc Test with L4 · 4-GPU" + source_file_dependencies: omni_qwen2_5_omni_doc + timeout_in_minutes: 90 + commands: + - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" + - pytest -s -v tests/examples/ -m "full_model and omni and L4 and B200 and cards_4" --run-level "full_model" + + - label: ":full_moon: Omni · Doc Test with H100 · 2-GPU" + source_file_dependencies: omni_qwen3_omni_doc + timeout_in_minutes: 90 + commands: + - pytest -s -v tests/examples/ -m "full_model and omni and H100 and B200 and cards_2" --run-level "full_model" + + - label: ":full_moon: Omni · Accuracy Test" + source_file_dependencies: omni_qwen3_omni_accuracy + timeout_in_minutes: 180 + artifact_paths: + - tests/e2e/accuracy/qwen3_omni/results/qwen_omni_acc/*.json + commands: + - export SEED_TTS_WER_EVAL=1 + - export SEED_TTS_EVAL_DEVICE=cuda:1 + - pytest -s -v tests/e2e/accuracy/qwen3_omni/test_qwen3_omni.py -m "full_model and H100 and B200 and cards_2" --run-level full_model + + - label: ":full_moon: Omni · MiniCPM-o 4.5 · Accuracy Test" + source_file_dependencies: omni_minicpmo_4_5_accuracy + timeout_in_minutes: 180 + artifact_paths: + - tests/e2e/accuracy/minicpmo_4_5/results/*.json + commands: + - export SEED_TTS_WER_EVAL=1 + - export SEED_TTS_EVAL_DEVICE=cuda:1 + - pytest -s -v tests/e2e/accuracy/minicpmo_4_5/test_minicpmo_4_5.py -m "full_model and H100 and B200 and cuda and (cards_1 or cards_2)" --run-level full_model + + - label: ":full_moon: Omni · Perf Test · No Async Chunk" + source_file_dependencies: omni_qwen3_omni_perf + key: nightly-omni-performance-no-async-chunk + timeout_in_minutes: 300 + artifact_paths: + - tests/dfx/perf/results/*.json + commands: + - export BENCHMARK_DIR=tests/dfx/perf/results + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_no_async_chunk.json -m "H100 and B200 and cards_2" + + - label: ":full_moon: Omni · Perf Test · Async Chunk" + source_file_dependencies: omni_qwen3_omni_perf + key: nightly-omni-performance-async-chunk + timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json + commands: + - export BENCHMARK_DIR=tests/dfx/perf/results + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen3_omni_async_chunk.json -m "H100 and B200 and full_model and cards_2" + + - label: ":full_moon: Omni · MiniCPM-o 4.5 · Perf Test" + source_file_dependencies: omni_minicpmo_4_5_perf + key: nightly-omni-performance-minicpmo-4-5 + timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json + commands: + - export BENCHMARK_DIR=tests/dfx/perf/results + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minicpmo_4_5.json -m "H100 and B200 and cards_1" + + - label: ":full_moon: Omni · MiniCPM-o 4.5 · Duplex Seed-TTS Perf Test" + source_file_dependencies: omni_minicpmo_4_5_duplex_perf + key: nightly-omni-performance-minicpmo-4-5-duplex-seed-tts + timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json + commands: + - export BENCHMARK_DIR=tests/dfx/perf/results + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minicpmo_4_5_duplex_seed_tts.json -m "H100 and B200 and cards_1" + + - label: ":full_moon: Omni · Multi-Replica Startup Test with 4x H100" + source_file_dependencies: omni_qwen3_omni_function + timeout_in_minutes: 45 + commands: + - pytest -s -v tests/e2e/online_serving/test_qwen3_omni_multi_replicas.py -m "full_model and H100 and B200 and cards_4" --run-level "core_model" - group: ":card_index_dividers: TTS Model Test" key: nightly-tts-test-group @@ -123,14 +122,14 @@ steps: if: build.env("NIGHTLY") == "1" || build.pull_request.labels includes "nightly-test" steps: -# - label: ":full_moon: TTS · Function Test with L4" -# source_file_dependencies: tts_qwen3_tts_function -# timeout_in_minutes: 120 -# commands: -# - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" -# - export VLLM_USE_DEEP_GEMM="0" -# - export VLLM_MOE_USE_DEEP_GEMM="0" -# - pytest -s -v tests/e2e/ -m "full_model and L4 and B200 and tts and cards_1" --run-level "full_model" --ignore=tests/e2e/accuracy + - label: ":full_moon: TTS · Function Test with L4" + source_file_dependencies: tts_qwen3_tts_function + timeout_in_minutes: 120 + commands: + - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" + - export VLLM_USE_DEEP_GEMM="0" + - export VLLM_MOE_USE_DEEP_GEMM="0" + - pytest -s -v tests/e2e/ -m "full_model and L4 and B200 and tts and cards_1" --run-level "full_model" --ignore=tests/e2e/accuracy - label: ":full_moon: TTS · Perf Test · Single-GPU" source_file_dependencies: tts_qwen3_tts_perf @@ -143,28 +142,28 @@ steps: - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_tts.json -m "H100 and B200 and cards_1" -# - label: ":full_moon: TTS · Perf Test · 2-GPU" -# source_file_dependencies: tts_qwen3_tts_perf -# key: nightly-tts-performance-2gpu -# timeout_in_minutes: 180 -# artifact_paths: -# - tests/dfx/perf/results/*.json -# commands: -# - export BENCHMARK_DIR=tests/dfx/perf/results -# - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" -# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_tts.json -m "H100 and B200 and cards_2" - -# - group: ":card_index_dividers: Tiny Model Tests [Multi-GPU]" -# key: nightly-tiny-model-test-group -# depends_on: upload-nightly-pipeline -# if: >- -# build.env("NIGHTLY") == "1" || -# build.pull_request.labels includes "nightly-test" -# steps: - -# - label: ":full_moon: Tiny Model Tests · Diffusion (Multi-GPU accelerations) · 2-GPU" -# source_file_dependencies: diffusion_tiny_model -# timeout_in_minutes: 30 + - label: ":full_moon: TTS · Perf Test · 2-GPU" + source_file_dependencies: tts_qwen3_tts_perf + key: nightly-tts-performance-2gpu + timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json + commands: + - export BENCHMARK_DIR=tests/dfx/perf/results + - export VLLM_ALLOW_LONG_MAX_MODEL_LEN="1" + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_tts.json -m "H100 and B200 and cards_2" + + - group: ":card_index_dividers: Tiny Model Tests [Multi-GPU]" + key: nightly-tiny-model-test-group + depends_on: upload-nightly-pipeline + if: >- + build.env("NIGHTLY") == "1" || + build.pull_request.labels includes "nightly-test" + steps: + + - label: ":full_moon: Tiny Model Tests · Diffusion (Multi-GPU accelerations) · 2-GPU" + source_file_dependencies: diffusion_tiny_model + timeout_in_minutes: 30 # NOTE: Single-GPU (core_model) tiny model tests run on the ready label, # so nightly only adds the multi-GPU parallelism configurations. # @@ -172,357 +171,357 @@ steps: # --run-level `core_model` is used to indicate that we should use tiny model weights # for the CI, since these tests should pass irrespective of whether or not we use # the full weights or not. -# commands: -# - pytest -sv tests/model_tests/diffusion -m 'full_model and L4 and B200 and cuda and cards_2' --run-level "core_model" + commands: + - pytest -sv tests/model_tests/diffusion -m 'full_model and L4 and B200 and cuda and cards_2' --run-level "core_model" -# - label: ":full_moon: Tiny Model Tests · Diffusion (Multi-GPU accelerations) · 4-GPU" -# source_file_dependencies: diffusion_tiny_model -# timeout_in_minutes: 30 + - label: ":full_moon: Tiny Model Tests · Diffusion (Multi-GPU accelerations) · 4-GPU" + source_file_dependencies: diffusion_tiny_model + timeout_in_minutes: 30 # CFG+TP extra_test_groups request 2×2=4 devices (see get_required_device_count). -# commands: -# - pytest -sv tests/model_tests/diffusion -m 'full_model and L4 and B200 and cuda and cards_4' --run-level "core_model" - -# - group: ":card_index_dividers: Diffusion X2I(&A&T) Model Test" -# key: nightly-diffusion-x2iat-group -# depends_on: upload-nightly-pipeline -# if: >- -# build.env("NIGHTLY") == "1" || -# build.pull_request.labels includes "nightly-test" -# steps: - -# - label: ":full_moon: Diffusion X2I(&A&T) · Function Test with H100 · Single-GPU" -# source_file_dependencies: diffusion_qwen_image_function -# timeout_in_minutes: 120 -# commands: -# - >- -# pytest -sv -# tests/e2e/online_serving/test_qwen_image_expansion.py -# -m "full_model and diffusion and H100 and B200 and cards_1" -# --run-level "full_model" - -# - label: ":robot_face: Robot Policy OpenPI · π0.5 · H100 · Single-GPU" -# timeout_in_minutes: 120 -# commands: -# - pip install --no-deps "openpi-client @ git+https://github.com/Physical-Intelligence/openpi.git@215abfb217dbac7d5f1273282331b9b1866c0479#subdirectory=packages/openpi-client" -# - >- -# pytest -sv -# tests/e2e/online_serving/test_pi05_expansion.py -# -m "full_model and diffusion and H100 and cards_1" -# --run-level "full_model" -# mirror_hardwares: h100_1 - -# - label: ":full_moon: Diffusion X2I(&A&T) · Function Test with H100 · 2-GPU" -# source_file_dependencies: diffusion_qwen_image_function -# timeout_in_minutes: 120 -# commands: -# - >- -# pytest -sv -# tests/e2e/online_serving/test_qwen_image_expansion.py -# -m "full_model and diffusion and H100 and B200 and cards_2" -# --run-level "full_model" - -# - label: ":full_moon: Diffusion X2I(&A&T) · Function Test with L4 · Single-GPU" -# source_file_dependencies: diffusion_qwen_image_function -# timeout_in_minutes: 120 -# commands: -# - pytest -sv tests/e2e/offline_inference/test_qwen_image_autoround_w4a16_expansion.py -m "full_model and diffusion and L4 and B200 and cards_1" --run-level "full_model" - - -# - label: ":full_moon: Diffusion X2I(&A&T) · Doc Test" -# source_file_dependencies: diffusion_text_to_image_doc -# timeout_in_minutes: 60 -# commands: -# - pytest -s -v tests/examples/*/test_text_to_image.py -m "full_model and example and H100 and B200 and (cards_1 or cards_2)" --run-level "full_model" - -# - label: ":full_moon: Diffusion X2I(&A&T) · Accuracy Test" -# source_file_dependencies: diffusion_qwen_image_accuracy -# timeout_in_minutes: 180 -# commands: -# - export VLLM_HTTP_TIMEOUT_KEEP_ALIVE=120 -# - pytest -s -v tests/e2e/accuracy/test_qwen_image.py -m "H100 and B200 and cards_1" --run-level full_model -# - pytest -s -v tests/e2e/accuracy/test_diffusers_backend_similarity.py -k '2i_matches_diffusers' -m "full_model and H100 and B200 and cards_1" --run-level full_model - -# - label: ":full_moon: HunyuanImage3-DIT · Accuracy Test" -# source_file_dependencies: diffusion_hunyuan_image3_accuracy -# timeout_in_minutes: 180 -# commands: -# - export HUNYUAN_IMAGE3_MODEL="tencent/HunyuanImage-3.0-Instruct" -# - export HUNYUAN_IMAGE3_DEVICES="0,1,2,3" -# - pytest -s -v tests/e2e/accuracy/test_hunyuan_image3_pixel_accuracy.py -m "full_model and H100 and B200 and cards_4" --run-level full_model - -# - label: ":full_moon: Diffusion X2I(&A&T) · Perf Test · Qwen-Image · Single-GPU" -# source_file_dependencies: diffusion_qwen_image_perf -# key: nightly-diffusion-x2iat-performance-qwen-image-single -# timeout_in_minutes: 180 -# artifact_paths: -# - tests/dfx/perf/results/diffusion_result_*.json -# - tests/dfx/perf/results/logs/*.log -# commands: -# - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results -# - export CACHE_DIT_VERSION=1.5.0 -# - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_1" - -# - label: ":full_moon: Diffusion X2I(&A&T) · Perf Test · Qwen-Image · 4-GPU" -# source_file_dependencies: diffusion_qwen_image_perf -# key: nightly-diffusion-x2iat-performance-qwen-image-4gpu -# timeout_in_minutes: 180 -# artifact_paths: -# - tests/dfx/perf/results/diffusion_result_*.json -# - tests/dfx/perf/results/logs/*.log -# commands: -# - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results -# - export CACHE_DIT_VERSION=1.5.0 + commands: + - pytest -sv tests/model_tests/diffusion -m 'full_model and L4 and B200 and cuda and cards_4' --run-level "core_model" + + - group: ":card_index_dividers: Diffusion X2I(&A&T) Model Test" + key: nightly-diffusion-x2iat-group + depends_on: upload-nightly-pipeline + if: >- + build.env("NIGHTLY") == "1" || + build.pull_request.labels includes "nightly-test" + steps: + + - label: ":full_moon: Diffusion X2I(&A&T) · Function Test with H100 · Single-GPU" + source_file_dependencies: diffusion_qwen_image_function + timeout_in_minutes: 120 + commands: + - >- + pytest -sv + tests/e2e/online_serving/test_qwen_image_expansion.py + -m "full_model and diffusion and H100 and B200 and cards_1" + --run-level "full_model" + + - label: ":robot_face: Robot Policy OpenPI · π0.5 · H100 · Single-GPU" + timeout_in_minutes: 120 + commands: + - pip install --no-deps "openpi-client @ git+https://github.com/Physical-Intelligence/openpi.git@215abfb217dbac7d5f1273282331b9b1866c0479#subdirectory=packages/openpi-client" + - >- + pytest -sv + tests/e2e/online_serving/test_pi05_expansion.py + -m "full_model and diffusion and H100 and cards_1" + --run-level "full_model" + mirror_hardwares: h100_1 + + - label: ":full_moon: Diffusion X2I(&A&T) · Function Test with H100 · 2-GPU" + source_file_dependencies: diffusion_qwen_image_function + timeout_in_minutes: 120 + commands: + - >- + pytest -sv + tests/e2e/online_serving/test_qwen_image_expansion.py + -m "full_model and diffusion and H100 and B200 and cards_2" + --run-level "full_model" + + - label: ":full_moon: Diffusion X2I(&A&T) · Function Test with L4 · Single-GPU" + source_file_dependencies: diffusion_qwen_image_function + timeout_in_minutes: 120 + commands: + - pytest -sv tests/e2e/offline_inference/test_qwen_image_autoround_w4a16_expansion.py -m "full_model and diffusion and L4 and B200 and cards_1" --run-level "full_model" + + + - label: ":full_moon: Diffusion X2I(&A&T) · Doc Test" + source_file_dependencies: diffusion_text_to_image_doc + timeout_in_minutes: 60 + commands: + - pytest -s -v tests/examples/*/test_text_to_image.py -m "full_model and example and H100 and B200 and (cards_1 or cards_2)" --run-level "full_model" + + - label: ":full_moon: Diffusion X2I(&A&T) · Accuracy Test" + source_file_dependencies: diffusion_qwen_image_accuracy + timeout_in_minutes: 180 + commands: + - export VLLM_HTTP_TIMEOUT_KEEP_ALIVE=120 + - pytest -s -v tests/e2e/accuracy/test_qwen_image.py -m "H100 and B200 and cards_1" --run-level full_model + - pytest -s -v tests/e2e/accuracy/test_diffusers_backend_similarity.py -k '2i_matches_diffusers' -m "full_model and H100 and B200 and cards_1" --run-level full_model + + - label: ":full_moon: HunyuanImage3-DIT · Accuracy Test" + source_file_dependencies: diffusion_hunyuan_image3_accuracy + timeout_in_minutes: 180 + commands: + - export HUNYUAN_IMAGE3_MODEL="tencent/HunyuanImage-3.0-Instruct" + - export HUNYUAN_IMAGE3_DEVICES="0,1,2,3" + - pytest -s -v tests/e2e/accuracy/test_hunyuan_image3_pixel_accuracy.py -m "full_model and H100 and B200 and cards_4" --run-level full_model + + - label: ":full_moon: Diffusion X2I(&A&T) · Perf Test · Qwen-Image · Single-GPU" + source_file_dependencies: diffusion_qwen_image_perf + key: nightly-diffusion-x2iat-performance-qwen-image-single + timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/diffusion_result_*.json + - tests/dfx/perf/results/logs/*.log + commands: + - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results + - export CACHE_DIT_VERSION=1.5.0 + - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_1" + + - label: ":full_moon: Diffusion X2I(&A&T) · Perf Test · Qwen-Image · 4-GPU" + source_file_dependencies: diffusion_qwen_image_perf + key: nightly-diffusion-x2iat-performance-qwen-image-4gpu + timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/diffusion_result_*.json + - tests/dfx/perf/results/logs/*.log + commands: + - export DIFFUSION_BENCHMARK_DIR=tests/dfx/perf/results + - export CACHE_DIT_VERSION=1.5.0 # Do not pin DIFFUSION_ATTENTION_BACKEND: H100 auto-selects FLASH_ATTN; # B200 rejects explicit FLASH_ATTN without FA4 and uses CUDNN/TRTLLM. -# - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_4" - -# - group: ":card_index_dividers: Diffusion X2V Model Test" -# key: nightly-diffusion-x2v-group -# depends_on: upload-nightly-pipeline -# if: >- -# build.env("NIGHTLY") == "1" || -# build.pull_request.labels includes "nightly-test" -# steps: - -# - label: ":full_moon: Diffusion X2V · Wan2.2 T2V Function Test · Single-GPU" -# source_file_dependencies: diffusion_wan22_function -# timeout_in_minutes: 90 -# commands: -# - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "t2v" -m "full_model and cuda and H100 and B200 and cards_1" --run-level "full_model" - -# - label: ":full_moon: Diffusion X2V · Wan2.2 T2V Function Test · 2-GPU" -# source_file_dependencies: diffusion_wan22_function -# timeout_in_minutes: 90 -# commands: -# - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "t2v" -m "full_model and cuda and H100 and B200 and cards_2" --run-level "full_model" - -# - label: ":full_moon: Diffusion X2V · Wan2.2 I2V/TI2V Function Test · Single-GPU" -# source_file_dependencies: diffusion_wan22_function -# timeout_in_minutes: 90 -# commands: -# - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "i2v" -m "full_model and cuda and H100 and B200 and cards_1" --run-level "full_model" - -# - label: ":full_moon: Diffusion X2V · Wan2.2 I2V/TI2V Function Test · 2-GPU" -# source_file_dependencies: diffusion_wan22_function -# timeout_in_minutes: 90 -# commands: -# - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "i2v" -m "full_model and cuda and H100 and B200 and cards_2" --run-level "full_model" - -# - label: ":full_moon: Diffusion X2V · HunyuanVideo-1.5 Function Test" -# source_file_dependencies: diffusion_hunyuan_video_function -# timeout_in_minutes: 90 -# commands: -# - pytest -s -v tests/e2e/online_serving/test_hunyuan_video_15_expansion.py -m "full_model and H100 and B200 and cuda and (cards_1 or cards_2)" --run-level "full_model" - -# - label: ":full_moon: Diffusion X2V · LingBot Function Test" -# source_file_dependencies: diffusion_lingbot_function -# timeout_in_minutes: 120 -# commands: -# - >- -# pytest -s -v -# tests/e2e/online_serving/test_lingbot_video.py -# tests/e2e/online_serving/test_lingbot_video_moe.py -# tests/e2e/offline_inference/test_lingbot_world_v2.py -# -m "full_model and diffusion and H100 and B200 and cards_1" -# --run-level "full_model" - -# - label: ":full_moon: Diffusion X2V · Doc Test" -# source_file_dependencies: diffusion_image_to_video_doc -# timeout_in_minutes: 60 -# commands: -# - pytest -s -v tests/examples/offline_inference/test_image_to_video.py -m "full_model and example and H100 and B200 and cards_1" --run-level "full_model" - -# - label: ":full_moon: Diffusion X2V · Wan2.2 Autoround Function Test" -# source_file_dependencies: diffusion_wan22_function -# timeout_in_minutes: 90 -# commands: -# - pytest -s -v tests/e2e/offline_inference/test_wan22_autoround_w4a16_expansion.py -m "full_model and H100 and B200 and cuda and (cards_1 or cards_2)" --run-level "full_model" - -# - label: ":full_moon: Diffusion X2V · Wan2.2 Device-Postprocess Equivalence Test" -# source_file_dependencies: diffusion_wan22_function -# timeout_in_minutes: 90 -# commands: -# - | -# VLLM_OMNI_DEVICE_POSTPROCESS_MODEL=Wan-AI/Wan2.2-TI2V-5B-Diffusers \ -# pytest -s -v tests/e2e/offline_inference/test_device_postprocess_equivalence.py -m "full_model and H100 and B200 and cuda and cards_1" --run-level "full_model" - -# - label: ":full_moon: Diffusion X2V · Wan2.2 Accuracy Test" -# source_file_dependencies: diffusion_wan22_accuracy -# timeout_in_minutes: 180 -# commands: -# - pytest -s -v tests/e2e/accuracy/wan22_i2v/test_wan22_i2v_video_similarity.py -m "full_model and H100 and B200 and (cards_1 or cards_2)" --run-level full_model -# - pytest -s -v tests/e2e/accuracy/test_diffusers_backend_similarity.py -k '2v_matches_diffusers' -m "full_model and H100 and B200 and cards_1" --run-level full_model - -# - label: ":full_moon: Diffusion X2V · MiniMax H3 I2VA/Ref2VA Accuracy Test" -# source_file_dependencies: diffusion_minimax_h3_accuracy -# timeout_in_minutes: 180 -# commands: -# - pytest -s -v tests/e2e/accuracy/minimax_h3/test_minimax_h3_i2va_ref2va_similarity.py -m "full_model and H100 and B200 and cards_4" --run-level full_model - -# - label: ":full_moon: Diffusion X2V · HunyuanVideo-1.5 Accuracy Test" -# source_file_dependencies: diffusion_hunyuan_video_accuracy -# timeout_in_minutes: 180 -# commands: -# - pytest -s -v tests/e2e/accuracy/hunyuanvideo15_t2v/test_hunyuanvideo15_t2v_video_similarity.py -m "full_model and H100 and B200 and (cards_1 or cards_2)" --run-level full_model -# - pytest -s -v tests/e2e/accuracy/hunyuanvideo15_i2v/test_hunyuanvideo15_i2v_video_similarity.py -m "full_model and H100 and B200 and (cards_1 or cards_2)" --run-level full_model - -# - label: ":full_moon: Diffusion X2V · Perf Test · Single-GPU" -# source_file_dependencies: diffusion_wan22_perf -# key: nightly-diffusion-x2v-performance-single -# timeout_in_minutes: 180 -# artifact_paths: -# - tests/dfx/perf/results/*.json -# commands: -# - export BENCHMARK_DIR=tests/dfx/perf/results -# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json -m "H100 and B200 and cards_1" - -# - label: ":full_moon: Diffusion X2V · Perf Test · 2-GPU" -# source_file_dependencies: -# - diffusion_wan22_perf -# - diffusion_cosmos3_perf -# key: nightly-diffusion-x2v-performance-2gpu -# timeout_in_minutes: 180 -# artifact_paths: -# - tests/dfx/perf/results/*.json -# commands: -# - export BENCHMARK_DIR=tests/dfx/perf/results -# - | -# set +e -# pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json -m "H100 and B200 and cards_2" -# EXIT1=$$? -# pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_cosmos3_vllm_omni.json -m "H100 and B200 and cards_2" -# EXIT2=$$? -# if [ $$EXIT1 -ne 0 ]; then exit $$EXIT1; fi -# exit $$EXIT2 - -# - label: ":full_moon: Diffusion X2V · Perf Test · HunyuanVideo-1.5" -# source_file_dependencies: diffusion_hunyuan_video_perf -# key: nightly-diffusion-x2v-performance-hunyuanvideo15 -# timeout_in_minutes: 180 -# artifact_paths: -# - tests/dfx/perf/results/*.json -# commands: -# - export BENCHMARK_DIR=tests/dfx/perf/results -# - | -# set +e -# pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_hunyuanvideo15_t2v_vllm_omni.json -m "H100 and B200 and (cards_1 or cards_2)" -# EXIT1=$$? -# pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_hunyuanvideo15_i2v_vllm_omni.json -m "H100 and B200 and (cards_1 or cards_2)" -# EXIT2=$$? -# if [ $$EXIT1 -ne 0 ]; then exit $$EXIT1; fi -# exit $$EXIT2 - -# - label: ":full_moon: Diffusion X2V · Perf Test · MiniMax-H3" -# source_file_dependencies: diffusion_minimax_h3_perf -# key: nightly-diffusion-x2v-performance-minimax-h3 -# timeout_in_minutes: 180 -# artifact_paths: -# - tests/dfx/perf/results/*.json -# commands: -# - export BENCHMARK_DIR=tests/dfx/perf/results -# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json -m "H100 and B200 and cards_4" - -# - label: ":full_moon: Diffusion X2V · Perf Test · LingBot" -# source_file_dependencies: diffusion_lingbot_perf -# key: nightly-diffusion-x2v-performance-lingbot -# timeout_in_minutes: 180 -# artifact_paths: -# - tests/dfx/perf/results/*.json -# commands: -# - export BENCHMARK_DIR=tests/dfx/perf/results -# - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json -m "H100 and B200 and cards_1" - -# - group: ":card_index_dividers: Diffusion Test" -# key: nightly-diffusion-test-group -# depends_on: upload-nightly-pipeline -# if: build.env("NIGHTLY") == "1" || build.pull_request.labels includes "nightly-test" -# steps: -# - label: ":full_moon: Diffusion · Distributed Test with L4 · 2-GPU" -# source_file_dependencies: diffusion_distributed_attention -# timeout_in_minutes: 60 -# commands: -# - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and L4 and B200 and cards_2' --run-level "full_model" - -# - label: ":full_moon: Diffusion · Distributed Test with L4 · 3-GPU" -# source_file_dependencies: diffusion_distributed_attention -# timeout_in_minutes: 60 -# commands: -# - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and L4 and B200 and cards_3' --run-level "full_model" - -# - label: ":full_moon: Diffusion · Distributed Test with L4 · 4-GPU" -# source_file_dependencies: diffusion_distributed_attention -# timeout_in_minutes: 60 -# commands: -# - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and L4 and B200 and cards_4' --run-level "full_model" - -# - label: ":full_moon: Diffusion · Distributed Test with H100 · 2-GPU" -# source_file_dependencies: diffusion_distributed_attention -# timeout_in_minutes: 60 -# commands: -# - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and H100 and B200 and cards_2' --run-level "full_model" - -# - label: ":full_moon: Diffusion · Distributed Test with H100 · 4-GPU" -# source_file_dependencies: diffusion_distributed_attention -# timeout_in_minutes: 60 -# commands: -# - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and H100 and B200 and cards_4' --run-level "full_model" - -# - label: ":full_moon: Diffusion · Offloader Test" -# source_file_dependencies: diffusion_offloader -# timeout_in_minutes: 60 -# commands: -# - pytest -sv tests/diffusion/offloader -m "full_model and L4 and B200 and cuda and cards_1" - -# - label: ":full_moon: Diffusion · Quantization Test with H100 · Single-GPU" -# source_file_dependencies: diffusion_quantization -# timeout_in_minutes: 60 -# commands: -# - pytest -sv tests/diffusion/quantization -m 'full_model and cuda and H100 and B200 and cards_1' --run-level "full_model" - -# - label: ":full_moon: Diffusion · MiniMax H3 FP8 Quality Test with H100 · 2-GPU" -# source_file_dependencies: diffusion_minimax_h3_function -# timeout_in_minutes: 60 -# commands: -# - pytest -sv tests/diffusion/models/minimax_h3/test_minimax_h3_quantization_quality.py -m 'full_model and cuda and H100 and B200 and cards_2' --run-level "full_model" - -# - label: ":full_moon: Diffusion · Quantization Test with L4" -# source_file_dependencies: diffusion_quantization -# timeout_in_minutes: 60 -# commands: -# - pip install "vllm-gguf-plugin==0.0.4" -# - pytest -sv tests/diffusion/quantization -m 'full_model and cuda and L4 and B200 and cards_1' --run-level "full_model" + - pytest -s -v tests/dfx/perf/scripts/run_diffusion_benchmark.py --test-config-file tests/dfx/perf/tests/test_qwen_image_vllm_omni.json -m "H100 and B200 and cards_4" + + - group: ":card_index_dividers: Diffusion X2V Model Test" + key: nightly-diffusion-x2v-group + depends_on: upload-nightly-pipeline + if: >- + build.env("NIGHTLY") == "1" || + build.pull_request.labels includes "nightly-test" + steps: + + - label: ":full_moon: Diffusion X2V · Wan2.2 T2V Function Test · Single-GPU" + source_file_dependencies: diffusion_wan22_function + timeout_in_minutes: 90 + commands: + - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "t2v" -m "full_model and cuda and H100 and B200 and cards_1" --run-level "full_model" + + - label: ":full_moon: Diffusion X2V · Wan2.2 T2V Function Test · 2-GPU" + source_file_dependencies: diffusion_wan22_function + timeout_in_minutes: 90 + commands: + - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "t2v" -m "full_model and cuda and H100 and B200 and cards_2" --run-level "full_model" + + - label: ":full_moon: Diffusion X2V · Wan2.2 I2V/TI2V Function Test · Single-GPU" + source_file_dependencies: diffusion_wan22_function + timeout_in_minutes: 90 + commands: + - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "i2v" -m "full_model and cuda and H100 and B200 and cards_1" --run-level "full_model" + + - label: ":full_moon: Diffusion X2V · Wan2.2 I2V/TI2V Function Test · 2-GPU" + source_file_dependencies: diffusion_wan22_function + timeout_in_minutes: 90 + commands: + - pytest -s -v tests/e2e/online_serving/test_wan22_expansion.py -k "i2v" -m "full_model and cuda and H100 and B200 and cards_2" --run-level "full_model" + + - label: ":full_moon: Diffusion X2V · HunyuanVideo-1.5 Function Test" + source_file_dependencies: diffusion_hunyuan_video_function + timeout_in_minutes: 90 + commands: + - pytest -s -v tests/e2e/online_serving/test_hunyuan_video_15_expansion.py -m "full_model and H100 and B200 and cuda and (cards_1 or cards_2)" --run-level "full_model" + + - label: ":full_moon: Diffusion X2V · LingBot Function Test" + source_file_dependencies: diffusion_lingbot_function + timeout_in_minutes: 120 + commands: + - >- + pytest -s -v + tests/e2e/online_serving/test_lingbot_video.py + tests/e2e/online_serving/test_lingbot_video_moe.py + tests/e2e/offline_inference/test_lingbot_world_v2.py + -m "full_model and diffusion and H100 and B200 and cards_1" + --run-level "full_model" + + - label: ":full_moon: Diffusion X2V · Doc Test" + source_file_dependencies: diffusion_image_to_video_doc + timeout_in_minutes: 60 + commands: + - pytest -s -v tests/examples/offline_inference/test_image_to_video.py -m "full_model and example and H100 and B200 and cards_1" --run-level "full_model" + + - label: ":full_moon: Diffusion X2V · Wan2.2 Autoround Function Test" + source_file_dependencies: diffusion_wan22_function + timeout_in_minutes: 90 + commands: + - pytest -s -v tests/e2e/offline_inference/test_wan22_autoround_w4a16_expansion.py -m "full_model and H100 and B200 and cuda and (cards_1 or cards_2)" --run-level "full_model" + + - label: ":full_moon: Diffusion X2V · Wan2.2 Device-Postprocess Equivalence Test" + source_file_dependencies: diffusion_wan22_function + timeout_in_minutes: 90 + commands: + - | + VLLM_OMNI_DEVICE_POSTPROCESS_MODEL=Wan-AI/Wan2.2-TI2V-5B-Diffusers \ + pytest -s -v tests/e2e/offline_inference/test_device_postprocess_equivalence.py -m "full_model and H100 and B200 and cuda and cards_1" --run-level "full_model" + + - label: ":full_moon: Diffusion X2V · Wan2.2 Accuracy Test" + source_file_dependencies: diffusion_wan22_accuracy + timeout_in_minutes: 180 + commands: + - pytest -s -v tests/e2e/accuracy/wan22_i2v/test_wan22_i2v_video_similarity.py -m "full_model and H100 and B200 and (cards_1 or cards_2)" --run-level full_model + - pytest -s -v tests/e2e/accuracy/test_diffusers_backend_similarity.py -k '2v_matches_diffusers' -m "full_model and H100 and B200 and cards_1" --run-level full_model + + - label: ":full_moon: Diffusion X2V · MiniMax H3 I2VA/Ref2VA Accuracy Test" + source_file_dependencies: diffusion_minimax_h3_accuracy + timeout_in_minutes: 180 + commands: + - pytest -s -v tests/e2e/accuracy/minimax_h3/test_minimax_h3_i2va_ref2va_similarity.py -m "full_model and H100 and B200 and cards_4" --run-level full_model + + - label: ":full_moon: Diffusion X2V · HunyuanVideo-1.5 Accuracy Test" + source_file_dependencies: diffusion_hunyuan_video_accuracy + timeout_in_minutes: 180 + commands: + - pytest -s -v tests/e2e/accuracy/hunyuanvideo15_t2v/test_hunyuanvideo15_t2v_video_similarity.py -m "full_model and H100 and B200 and (cards_1 or cards_2)" --run-level full_model + - pytest -s -v tests/e2e/accuracy/hunyuanvideo15_i2v/test_hunyuanvideo15_i2v_video_similarity.py -m "full_model and H100 and B200 and (cards_1 or cards_2)" --run-level full_model + + - label: ":full_moon: Diffusion X2V · Perf Test · Single-GPU" + source_file_dependencies: diffusion_wan22_perf + key: nightly-diffusion-x2v-performance-single + timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json + commands: + - export BENCHMARK_DIR=tests/dfx/perf/results + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json -m "H100 and B200 and cards_1" + + - label: ":full_moon: Diffusion X2V · Perf Test · 2-GPU" + source_file_dependencies: + - diffusion_wan22_perf + - diffusion_cosmos3_perf + key: nightly-diffusion-x2v-performance-2gpu + timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json + commands: + - export BENCHMARK_DIR=tests/dfx/perf/results + - | + set +e + pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_wan22_i2v_vllm_omni.json -m "H100 and B200 and cards_2" + EXIT1=$$? + pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_cosmos3_vllm_omni.json -m "H100 and B200 and cards_2" + EXIT2=$$? + if [ $$EXIT1 -ne 0 ]; then exit $$EXIT1; fi + exit $$EXIT2 + + - label: ":full_moon: Diffusion X2V · Perf Test · HunyuanVideo-1.5" + source_file_dependencies: diffusion_hunyuan_video_perf + key: nightly-diffusion-x2v-performance-hunyuanvideo15 + timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json + commands: + - export BENCHMARK_DIR=tests/dfx/perf/results + - | + set +e + pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_hunyuanvideo15_t2v_vllm_omni.json -m "H100 and B200 and (cards_1 or cards_2)" + EXIT1=$$? + pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_hunyuanvideo15_i2v_vllm_omni.json -m "H100 and B200 and (cards_1 or cards_2)" + EXIT2=$$? + if [ $$EXIT1 -ne 0 ]; then exit $$EXIT1; fi + exit $$EXIT2 + + - label: ":full_moon: Diffusion X2V · Perf Test · MiniMax-H3" + source_file_dependencies: diffusion_minimax_h3_perf + key: nightly-diffusion-x2v-performance-minimax-h3 + timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json + commands: + - export BENCHMARK_DIR=tests/dfx/perf/results + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minimax_h3_vllm_omni.json -m "H100 and B200 and cards_4" + + - label: ":full_moon: Diffusion X2V · Perf Test · LingBot" + source_file_dependencies: diffusion_lingbot_perf + key: nightly-diffusion-x2v-performance-lingbot + timeout_in_minutes: 180 + artifact_paths: + - tests/dfx/perf/results/*.json + commands: + - export BENCHMARK_DIR=tests/dfx/perf/results + - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_lingbot_video_vllm_omni.json -m "H100 and B200 and cards_1" + + - group: ":card_index_dividers: Diffusion Test" + key: nightly-diffusion-test-group + depends_on: upload-nightly-pipeline + if: build.env("NIGHTLY") == "1" || build.pull_request.labels includes "nightly-test" + steps: + - label: ":full_moon: Diffusion · Distributed Test with L4 · 2-GPU" + source_file_dependencies: diffusion_distributed_attention + timeout_in_minutes: 60 + commands: + - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and L4 and B200 and cards_2' --run-level "full_model" + + - label: ":full_moon: Diffusion · Distributed Test with L4 · 3-GPU" + source_file_dependencies: diffusion_distributed_attention + timeout_in_minutes: 60 + commands: + - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and L4 and B200 and cards_3' --run-level "full_model" + + - label: ":full_moon: Diffusion · Distributed Test with L4 · 4-GPU" + source_file_dependencies: diffusion_distributed_attention + timeout_in_minutes: 60 + commands: + - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and L4 and B200 and cards_4' --run-level "full_model" + + - label: ":full_moon: Diffusion · Distributed Test with H100 · 2-GPU" + source_file_dependencies: diffusion_distributed_attention + timeout_in_minutes: 60 + commands: + - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and H100 and B200 and cards_2' --run-level "full_model" + + - label: ":full_moon: Diffusion · Distributed Test with H100 · 4-GPU" + source_file_dependencies: diffusion_distributed_attention + timeout_in_minutes: 60 + commands: + - pytest -sv tests/diffusion/distributed/ -m 'full_model and cuda and H100 and B200 and cards_4' --run-level "full_model" + + - label: ":full_moon: Diffusion · Offloader Test" + source_file_dependencies: diffusion_offloader + timeout_in_minutes: 60 + commands: + - pytest -sv tests/diffusion/offloader -m "full_model and L4 and B200 and cuda and cards_1" + + - label: ":full_moon: Diffusion · Quantization Test with H100 · Single-GPU" + source_file_dependencies: diffusion_quantization + timeout_in_minutes: 60 + commands: + - pytest -sv tests/diffusion/quantization -m 'full_model and cuda and H100 and B200 and cards_1' --run-level "full_model" + + - label: ":full_moon: Diffusion · MiniMax H3 FP8 Quality Test with H100 · 2-GPU" + source_file_dependencies: diffusion_minimax_h3_function + timeout_in_minutes: 60 + commands: + - pytest -sv tests/diffusion/models/minimax_h3/test_minimax_h3_quantization_quality.py -m 'full_model and cuda and H100 and B200 and cards_2' --run-level "full_model" + + - label: ":full_moon: Diffusion · Quantization Test with L4" + source_file_dependencies: diffusion_quantization + timeout_in_minutes: 60 + commands: + - pip install "vllm-gguf-plugin==0.0.4" + - pytest -sv tests/diffusion/quantization -m 'full_model and cuda and L4 and B200 and cards_1' --run-level "full_model" # NOTE: ":email: Nightly Collection & Email" is deprecated and will be removed soon. # Do not add new depends_on entries or extend this step. -# - label: ":email: Nightly Collection & Email" -# key: nightly-perf-distribution -# depends_on: -# - nightly-omni-performance-no-async-chunk -# - nightly-omni-performance-async-chunk -# - nightly-tts-performance-single -# - nightly-tts-performance-2gpu -# - nightly-diffusion-x2iat-performance-qwen-image-single -# - nightly-diffusion-x2iat-performance-qwen-image-4gpu -# - nightly-diffusion-x2v-performance-single -# - nightly-diffusion-x2v-performance-2gpu -# if: build.env("NIGHTLY") == "1" && build.env("EMAIL_DISTRIBUTION") == "1" -# artifact_paths: -# - tests/dfx/perf/results/*.xlsx -# - tests/dfx/perf/results/*.html -# commands: -# - pip install openpyxl -# - export DEFAULT_INPUT_DIR=tests/dfx/perf/results -# - export DEFAULT_OUTPUT_DIR=tests/dfx/perf/results -# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-tts-performance-single -# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-tts-performance-2gpu -# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-omni-performance-no-async-chunk -# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-omni-performance-async-chunk -# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2iat-performance-qwen-image-single -# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2iat-performance-qwen-image-4gpu -# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2v-performance-single -# - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2v-performance-2gpu -# - python tools/nightly/generate_nightly_perf_excel.py -# - python tools/nightly/generate_nightly_perf_html.py -# - python tools/nightly/send_nightly_email.py --report-file "tests/dfx/perf/results/*.xlsx, tests/dfx/perf/results/*.html" -# agents: -# queue: "cpu_queue_premerge" + - label: ":email: Nightly Collection & Email" + key: nightly-perf-distribution + depends_on: + - nightly-omni-performance-no-async-chunk + - nightly-omni-performance-async-chunk + - nightly-tts-performance-single + - nightly-tts-performance-2gpu + - nightly-diffusion-x2iat-performance-qwen-image-single + - nightly-diffusion-x2iat-performance-qwen-image-4gpu + - nightly-diffusion-x2v-performance-single + - nightly-diffusion-x2v-performance-2gpu + if: build.env("NIGHTLY") == "1" && build.env("EMAIL_DISTRIBUTION") == "1" + artifact_paths: + - tests/dfx/perf/results/*.xlsx + - tests/dfx/perf/results/*.html + commands: + - pip install openpyxl + - export DEFAULT_INPUT_DIR=tests/dfx/perf/results + - export DEFAULT_OUTPUT_DIR=tests/dfx/perf/results + - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-tts-performance-single + - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-tts-performance-2gpu + - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-omni-performance-no-async-chunk + - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-omni-performance-async-chunk + - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2iat-performance-qwen-image-single + - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2iat-performance-qwen-image-4gpu + - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2v-performance-single + - buildkite-agent artifact download "tests/dfx/perf/results/*.json" . --step nightly-diffusion-x2v-performance-2gpu + - python tools/nightly/generate_nightly_perf_excel.py + - python tools/nightly/generate_nightly_perf_html.py + - python tools/nightly/send_nightly_email.py --report-file "tests/dfx/perf/results/*.xlsx, tests/dfx/perf/results/*.html" + agents: + queue: "cpu_queue_premerge" From 71db77bb55ac3a42b636dfb70fe1f248f55ef6ab Mon Sep 17 00:00:00 2001 From: wangyu <410167048@qq.com> Date: Sun, 20 Sep 2026 17:01:41 +0800 Subject: [PATCH 13/15] Enhance benchmark parameter handling and image/video reference processing - Added `benchmark_params_name` to `run_benchmark` for distinct identification of benchmark configurations in results. - Updated tests to validate the persistence of distinct benchmark parameters under the same test name. - Refactored image and video reference input functions to yield structured references, ensuring compatibility with form data handling. These improvements enhance the clarity and functionality of the benchmarking framework, particularly in managing multimodal content references. Signed-off-by: wangyu <410167048@qq.com> --- tests/benchmarks/patch/test_patch.py | 87 +++++++++++++++++++- tests/dfx/conftest.py | 30 ++++++- tests/dfx/perf/scripts/run_benchmark.py | 2 + tests/dfx/perf/tests/test_runner_metadata.py | 84 +++++++++++++++++-- tools/nightly/generate_nightly_perf_excel.py | 19 ++++- tools/nightly/generate_nightly_perf_html.py | 11 ++- vllm_omni/benchmarks/patch/patch.py | 36 +++++--- 7 files changed, 240 insertions(+), 29 deletions(-) diff --git a/tests/benchmarks/patch/test_patch.py b/tests/benchmarks/patch/test_patch.py index 6926e39d19d..b2a274e9bd0 100644 --- a/tests/benchmarks/patch/test_patch.py +++ b/tests/benchmarks/patch/test_patch.py @@ -30,6 +30,7 @@ _attach_seed_tts_to_request_func_input, _build_benchmark_session, _extract_stage_durations_from_payload, + _iter_image_reference_inputs, _iter_video_reference_inputs, _omni_request_timeout_s, async_request_openai_chat_omni_completions, @@ -1547,8 +1548,52 @@ def tracking_add_field(self, name, value=None, **kwargs): assert json.loads(payload) == reference +def test_image_reference_urls_from_random_mm_content(mocker: MockerFixture) -> None: + """random-mm image_url parts keep an explicit image_reference type.""" + import aiohttp + + content = [ + { + "type": "image_url", + "image_url": {"url": "https://example.com/ref.png"}, + }, + { + "type": "image_url", + "image_url": {"url": f"data:image/png;base64,{_MIN_PNG_B64}"}, + }, + ] + refs = list(_iter_image_reference_inputs(content)) + assert refs == [ + {"image_url": "https://example.com/ref.png"}, + {"image_url": f"data:image/png;base64,{_MIN_PNG_B64}"}, + ] + + captured: list[tuple[str, object]] = [] + real_add_field = aiohttp.FormData.add_field + + def tracking_add_field(self, name, value=None, **kwargs): + captured.append((str(name), value)) + return real_add_field(self, name, value, **kwargs) + + mocker.patch.object(aiohttp.FormData, "add_field", tracking_add_field) + form = aiohttp.FormData() + assert _add_video_reference_to_form(form, refs[0]) is True + assert _add_video_reference_to_form(form, refs[1]) is True + + field_names = [name for name, _ in captured] + assert field_names.count("image_reference") == 2 + assert "video_reference" not in field_names + payloads = [] + for name, value in captured: + if name != "image_reference": + continue + assert isinstance(value, (str, bytes, bytearray)) + payloads.append(json.loads(value)) + assert payloads == refs + + def test_video_reference_urls_from_random_mm_content(mocker: MockerFixture) -> None: - """random-mm video_url parts use the same form helper as image_reference.""" + """random-mm data:video_url parts upload via input_references.""" import aiohttp content = [ @@ -1557,8 +1602,8 @@ def test_video_reference_urls_from_random_mm_content(mocker: MockerFixture) -> N "video_url": {"url": "data:video/mp4;base64,AAAA"}, } ] - urls = list(_iter_video_reference_inputs(content)) - assert urls == ["data:video/mp4;base64,AAAA"] + refs = list(_iter_video_reference_inputs(content)) + assert refs == [{"video_url": "data:video/mp4;base64,AAAA"}] captured: list[tuple[str, object]] = [] real_add_field = aiohttp.FormData.add_field @@ -1569,11 +1614,45 @@ def tracking_add_field(self, name, value=None, **kwargs): mocker.patch.object(aiohttp.FormData, "add_field", tracking_add_field) form = aiohttp.FormData() - assert _add_video_reference_to_form(form, urls[0]) is True + assert _add_video_reference_to_form(form, refs[0]) is True uploaded = next(value for name, value in captured if name == "input_references") assert uploaded == base64.b64decode("AAAA") assert "input_reference" not in [name for name, _ in captured] assert "video_reference" not in [name for name, _ in captured] + assert "image_reference" not in [name for name, _ in captured] + + +def test_video_reference_https_url_from_random_mm_content(mocker: MockerFixture) -> None: + """HTTP(S) video_url parts must stay on video_reference, not image_reference.""" + import aiohttp + + content = [ + { + "type": "video_url", + "video_url": {"url": "https://example.com/ref.mp4"}, + } + ] + refs = list(_iter_video_reference_inputs(content)) + assert refs == [{"video_url": "https://example.com/ref.mp4"}] + + captured: list[tuple[str, object]] = [] + real_add_field = aiohttp.FormData.add_field + + def tracking_add_field(self, name, value=None, **kwargs): + captured.append((str(name), value)) + return real_add_field(self, name, value, **kwargs) + + mocker.patch.object(aiohttp.FormData, "add_field", tracking_add_field) + form = aiohttp.FormData() + assert _add_video_reference_to_form(form, refs[0]) is True + + field_names = [name for name, _ in captured] + assert field_names.count("video_reference") == 1 + assert "image_reference" not in field_names + assert "input_references" not in field_names + payload = next(value for name, value in captured if name == "video_reference") + assert isinstance(payload, (str, bytes, bytearray)) + assert json.loads(payload) == {"video_url": "https://example.com/ref.mp4"} def test_video_unsupported_image_reference_raises() -> None: diff --git a/tests/dfx/conftest.py b/tests/dfx/conftest.py index f67d2fe225d..2015f6389b7 100644 --- a/tests/dfx/conftest.py +++ b/tests/dfx/conftest.py @@ -660,6 +660,7 @@ def run_benchmark( random_output_len: Any | None = None, resource_label: str | None = None, num_warmups: int = 2, + benchmark_params_name: str | None = None, ) -> dict[str, Any]: """Run one ``vllm bench serve --omni`` iteration and return parsed metrics. @@ -668,15 +669,35 @@ def run_benchmark( maps keep every hardware bucket, but each metric is reduced to the value for this concurrency / request-rate step. If the benchmark exits without writing a result file, ``result_omni_template.json`` is used as a fallback. + + ``benchmark_params_name`` (the ``name`` field from ``benchmark_params``) is + persisted in the result JSON and filename so multiple generation configs + under one ``test_name`` stay distinguishable (e.g. Wan USP2 832x480 vs + 1280x720). """ current_dt = datetime.now().strftime("%Y%m%d-%H%M%S") ri = _safe_filename_token(random_input_len) ro = _safe_filename_token(random_output_len) hw = resource_label_for_filename(resource_label) + name_token = _safe_filename_token(benchmark_params_name) if benchmark_params_name else "" + if name_token == "na": + name_token = "" + name_parts = [f"result_{test_name}"] if hw: - result_filename = f"result_{test_name}_{hw}_{dataset_name}_{flow}_{num_prompt}_in{ri}_out{ro}_{current_dt}.json" - else: - result_filename = f"result_{test_name}_{dataset_name}_{flow}_{num_prompt}_in{ri}_out{ro}_{current_dt}.json" + name_parts.append(hw) + if name_token: + name_parts.append(name_token) + name_parts.extend( + [ + str(dataset_name), + str(flow), + str(num_prompt), + f"in{ri}", + f"out{ro}", + current_dt, + ] + ) + result_filename = "_".join(name_parts) + ".json" if "--result-filename" in args: print(f"The result file will be overwritten by {result_filename}") command = ( @@ -746,6 +767,9 @@ def _forward_stream(stream) -> None: if random_output_len is not None: result["random_output_len"] = random_output_len result["Hardware"] = hardware_json_value(resource_label) + result["test_name"] = test_name + if benchmark_params_name: + result["name"] = benchmark_params_name with open(result_path, "w", encoding="utf-8") as f: json.dump(result, f, ensure_ascii=False, indent=2) return result diff --git a/tests/dfx/perf/scripts/run_benchmark.py b/tests/dfx/perf/scripts/run_benchmark.py index 97239988dfd..5648ffff5f2 100644 --- a/tests/dfx/perf/scripts/run_benchmark.py +++ b/tests/dfx/perf/scripts/run_benchmark.py @@ -401,6 +401,7 @@ def to_list(value, default=None): random_output_len=params.get("random_output_len"), resource_label=resource_label, num_warmups=_resolve_num_warmups(params, default=2), + benchmark_params_name=params.get("name") if isinstance(params.get("name"), str) else None, ) assert_result(result, params, num_prompt) @@ -419,5 +420,6 @@ def to_list(value, default=None): random_output_len=params.get("random_output_len"), resource_label=resource_label, num_warmups=_resolve_num_warmups(params, default=max(2, int(concurrency))), + benchmark_params_name=params.get("name") if isinstance(params.get("name"), str) else None, ) assert_result(result, params, num_prompt) diff --git a/tests/dfx/perf/tests/test_runner_metadata.py b/tests/dfx/perf/tests/test_runner_metadata.py index 68717a91768..bdbd83df0a3 100644 --- a/tests/dfx/perf/tests/test_runner_metadata.py +++ b/tests/dfx/perf/tests/test_runner_metadata.py @@ -341,12 +341,84 @@ def fake_start(server_param): finally: active_context.close() - assert events == [ - ("start", omni_p0_server), - ("stop", omni_p0_server), - ("start", tts_server), - ("stop", tts_server), - ] + +def test_run_benchmark_persists_distinct_benchmark_params_name(tmp_path, monkeypatch): + """Two benchmark_params under one test_name must keep distinct saved identity.""" + import io + from pathlib import Path + + from tests.dfx import conftest as dfx_conftest + + result_dir = tmp_path / "results" + result_dir.mkdir() + monkeypatch.setenv("BENCHMARK_DIR", str(result_dir)) + + class _FakePopen: + def __init__(self, command, **kwargs): + self.stdout = io.StringIO("") + self.stderr = io.StringIO("") + result_dir_idx = command.index("--result-dir") + filename_idx = command.index("--result-filename") + out = Path(command[result_dir_idx + 1]) / command[filename_idx + 1] + out.parent.mkdir(parents=True, exist_ok=True) + out.write_text(json.dumps({"completed": 3, "request_throughput": 0.01}), encoding="utf-8") + + def wait(self): + return 0 + + monkeypatch.setattr(dfx_conftest.subprocess, "Popen", _FakePopen) + + shared_test = "test_wan22_i2v_usp2" + name_a = "832x480_frames81_steps4" + name_b = "1280x720_frames121_steps4" + + result_a = dfx_conftest.run_benchmark( + args=["--host", "127.0.0.1", "--port", "8000"], + test_name=shared_test, + flow=1, + dataset_name="random-mm", + num_prompt=10, + random_input_len=8, + random_output_len=1, + resource_label="H800", + benchmark_params_name=name_a, + ) + result_b = dfx_conftest.run_benchmark( + args=["--host", "127.0.0.1", "--port", "8000"], + test_name=shared_test, + flow=1, + dataset_name="random-mm", + num_prompt=10, + random_input_len=8, + random_output_len=1, + resource_label="H800", + benchmark_params_name=name_b, + ) + + assert result_a["test_name"] == shared_test + assert result_b["test_name"] == shared_test + assert result_a["name"] == name_a + assert result_b["name"] == name_b + assert "benchmark_params" not in result_a + assert "benchmark_params" not in result_b + + files = sorted(p.name for p in result_dir.glob("result_*.json")) + assert len(files) == 2 + assert any(name_a in name for name in files) + assert any(name_b in name for name in files) + assert files[0] != files[1] + + def omni_group_key(record: dict) -> tuple: + return ( + record.get("model_id") or "", + record.get("test_name") or "", + record.get("name") or "", + record.get("dataset_name") or "", + record.get("max_concurrency") if record.get("max_concurrency") is not None else 0, + record.get("num_prompts") if record.get("num_prompts") is not None else 0, + ) + + assert omni_group_key(result_a) != omni_group_key(result_b) def test_is_hardware_nested_baseline(): diff --git a/tools/nightly/generate_nightly_perf_excel.py b/tools/nightly/generate_nightly_perf_excel.py index 54c4b3f3406..eae463173a0 100644 --- a/tools/nightly/generate_nightly_perf_excel.py +++ b/tools/nightly/generate_nightly_perf_excel.py @@ -1,4 +1,7 @@ #!/usr/bin/env python3 +# SPDX-License-Identifier: Apache-2.0 +# SPDX-FileCopyrightText: Copyright contributors to the vLLM-Omni project + """ Generate a nightly Excel performance report from JSON results. @@ -137,6 +140,7 @@ def _omni_group_key(record: dict[str, Any]) -> tuple[Any, ...]: return ( record.get("model_id") or "", record.get("test_name") or "", + record.get("name") or "", record.get("dataset_name") or "", record.get("max_concurrency") if record.get("max_concurrency") is not None else 0, record.get("num_prompts") if record.get("num_prompts") is not None else 0, @@ -171,6 +175,7 @@ def _load_summary_columns(script_dir: str) -> list[str]: "model_id", "tokenizer_id", "test_name", + "name", "dataset_name", "num_prompts", "request_rate", @@ -218,7 +223,7 @@ def _load_summary_columns(script_dir: str) -> list[str]: def _ensure_omni_summary_columns(summary_columns: list[str]) -> list[str]: """Ensure omni summary contains required columns, even when a custom columns file exists.""" - required = ("test_name", "dataset_name", "source_file") + required = ("test_name", "name", "dataset_name", "source_file") existing = set(summary_columns) if all(c in existing for c in required): return summary_columns @@ -232,8 +237,18 @@ def _ensure_omni_summary_columns(summary_columns: list[str]) -> list[str]: else: out.append("test_name") existing = set(out) - if "dataset_name" not in existing: + if "name" not in existing: if "test_name" in out: + idx = out.index("test_name") + 1 + out.insert(idx, "name") + else: + out.append("name") + existing = set(out) + if "dataset_name" not in existing: + if "name" in out: + idx = out.index("name") + 1 + out.insert(idx, "dataset_name") + elif "test_name" in out: idx = out.index("test_name") + 1 out.insert(idx, "dataset_name") else: diff --git a/tools/nightly/generate_nightly_perf_html.py b/tools/nightly/generate_nightly_perf_html.py index 11e9b0961ee..cb29f627c57 100644 --- a/tools/nightly/generate_nightly_perf_html.py +++ b/tools/nightly/generate_nightly_perf_html.py @@ -1,4 +1,7 @@ #!/usr/bin/env python3 +# SPDX-License-Identifier: Apache-2.0 +# SPDX-FileCopyrightText: Copyright contributors to the vLLM-Omni project + """ Generate an improved nightly HTML performance dashboard without modifying the existing generator. @@ -256,6 +259,7 @@ def _sort_omni_records(records: list[dict[str, Any]]) -> list[dict[str, Any]]: key=lambda r: ( r.get("model_id") or "", r.get("test_name") or "", + r.get("name") or "", r.get("dataset_name") or "", r.get("max_concurrency") or 0, r.get("num_prompts") or 0, @@ -1236,7 +1240,7 @@ def _build_html_document( const extraKey = prefix === "diff" ? "" : "tokenizer_id"; const metaKeys = prefix === "diff" ? ["test_name"] - : ["test_name", "max_concurrency", "num_prompts"]; + : ["test_name", "name", "max_concurrency", "num_prompts"]; const configFields = prefix === "diff" ? ["test_name", "model", "endpoint", "dataset"] : [ @@ -1245,13 +1249,14 @@ def _build_html_document( "model_id", "tokenizer_id", "test_name", + "name", "dataset_name", "max_concurrency", "num_prompts", ]; const labelFields = prefix === "diff" ? ["test_name", "dataset"] - : ["test_name", "dataset_name"]; + : ["test_name", "name", "dataset_name"]; const latestMetric = prefix === "diff" ? "throughput_qps" : "output_throughput"; const secondaryMetric = prefix === "diff" ? "latency_mean" : "mean_e2el_ms"; const modelList = document.getElementById(`${{prefix}}-model-list`); @@ -1650,7 +1655,7 @@ def _build_html_document( '
Chat/audio benchmark history with ' "trend charts grouped by throughput, latency, and audio metrics." '
' - "Grouping test_name + dataset + concurrency + prompts
" + "Grouping test_name + name + dataset + concurrency + prompts
" '
Snapshots single-point series ' "stay readable
" ), diff --git a/vllm_omni/benchmarks/patch/patch.py b/vllm_omni/benchmarks/patch/patch.py index b61c94c6db2..fa32c59ada0 100644 --- a/vllm_omni/benchmarks/patch/patch.py +++ b/vllm_omni/benchmarks/patch/patch.py @@ -786,7 +786,13 @@ def _guess_mime_type(path: str) -> str: def _iter_image_reference_inputs(value: Any) -> Iterable[Any]: - """Yield image references from benchmark multimodal content.""" + """Yield image references from benchmark multimodal content. + + ``random-mm`` image buckets arrive as OpenAI chat parts + ``{"type": "image_url", "image_url": {"url": ...}}``. Yield + ``{"image_url": url}`` so the form helper keeps an explicit image + type (symmetric with ``_iter_video_reference_inputs``). + """ if value is None: return if isinstance(value, list): @@ -802,10 +808,10 @@ def _iter_image_reference_inputs(value: Any) -> Iterable[Any]: image_url = value.get("image_url") if isinstance(image_url, dict): url = image_url.get("url") - if url: - yield url - elif image_url: - yield image_url + if isinstance(url, str) and url: + yield {"image_url": url} + elif isinstance(image_url, str) and image_url: + yield {"image_url": image_url} return for key in ("image", "images"): @@ -813,12 +819,14 @@ def _iter_image_reference_inputs(value: Any) -> Iterable[Any]: yield from _iter_image_reference_inputs(value[key]) -def _iter_video_reference_inputs(value: Any) -> Iterable[str]: - """Yield video references from benchmark multimodal content. +def _iter_video_reference_inputs(value: Any) -> Iterable[dict[str, str]]: + """Yield structured video references from benchmark multimodal content. ``random-mm`` video buckets arrive as OpenAI chat parts - ``{"type": "video_url", "video_url": {"url": ...}}``. The videos API - expects ``video_reference`` with a string ``video_url``. + ``{"type": "video_url", "video_url": {"url": ...}}``. Yield + ``{"video_url": url}`` so ``_add_video_reference_to_form`` keeps the + video branch (HTTP(S) bare strings would otherwise become + ``image_reference``). """ if value is None: return @@ -834,9 +842,9 @@ def _iter_video_reference_inputs(value: Any) -> Iterable[str]: if isinstance(video_url, dict): url = video_url.get("url") if isinstance(url, str) and url: - yield url + yield {"video_url": url} elif isinstance(video_url, str) and video_url: - yield video_url + yield {"video_url": video_url} return for key in ("video", "videos"): @@ -854,6 +862,12 @@ def _add_image_edit_input_to_form(form: aiohttp.FormData, image_input: Any) -> N ) return + if isinstance(image_input, Mapping) and _is_structured_image_reference(image_input): + image_url = image_input.get("image_url") + if isinstance(image_url, str) and image_url: + _add_image_edit_input_to_form(form, image_url) + return + if isinstance(image_input, str): if image_input.startswith(("data:image", "http://", "https://")): form.add_field("url", image_input) From d1b54f12df1b12714c0da967dea5b9e54f894e61 Mon Sep 17 00:00:00 2001 From: wangyu <410167048@qq.com> Date: Mon, 21 Sep 2026 11:32:05 +0800 Subject: [PATCH 14/15] Refactor video reference handling and enhance test coverage - Introduced `_add_combined_video_form_references` to streamline the addition of image and video references in form data, ensuring compatibility with server requirements. - Updated `_add_video_reference_to_form` to reject unsupported `file_id` references, improving error handling. - Added new tests to validate the rejection of `file_id` references and ensure correct processing of combined image and video references. These changes enhance the robustness and clarity of video and image reference handling in the benchmarking framework. Signed-off-by: wangyu <410167048@qq.com> --- tests/benchmarks/patch/test_patch.py | 49 ++++++++++++++++- vllm_omni/benchmarks/patch/patch.py | 78 +++++++++++++++++----------- 2 files changed, 94 insertions(+), 33 deletions(-) diff --git a/tests/benchmarks/patch/test_patch.py b/tests/benchmarks/patch/test_patch.py index b2a274e9bd0..37410b65af0 100644 --- a/tests/benchmarks/patch/test_patch.py +++ b/tests/benchmarks/patch/test_patch.py @@ -22,6 +22,7 @@ ) from vllm_omni.benchmarks.patch.patch import ( MixRequestFuncOutput, + _add_combined_video_form_references, _add_video_extra_body_to_form, _add_video_reference_to_form, _apply_image_metrics_from_payload, @@ -1515,8 +1516,7 @@ def tracking_add_field(self, name, value=None, **kwargs): "reference", [ {"image_url": "https://example.com/ref.png"}, - {"file_id": "file-abc"}, - [{"image_url": "https://example.com/a.png"}, {"file_id": "file-xyz"}], + [{"image_url": "https://example.com/a.png"}, {"image_url": "https://example.com/b.png"}], ], ) def test_video_structured_image_reference_serialized_to_form(reference: object, mocker: MockerFixture) -> None: @@ -1548,6 +1548,19 @@ def tracking_add_field(self, name, value=None, **kwargs): assert json.loads(payload) == reference +def test_video_file_id_reference_is_rejected() -> None: + """file_id is unsupported on the server and must not be sent as image_reference.""" + import aiohttp + + form = aiohttp.FormData() + with pytest.raises(ValueError, match="file_id is not supported yet"): + _add_video_reference_to_form(form, {"file_id": "file-abc"}) + with pytest.raises(ValueError, match="file_id is not supported yet"): + _add_video_reference_to_form(form, [{"image_url": "https://example.com/a.png"}, {"file_id": "file-xyz"}]) + with pytest.raises(ValueError, match="file_id is not supported yet"): + _add_combined_video_form_references(form, None, {"video_reference": {"file_id": "file-vid"}}) + + def test_image_reference_urls_from_random_mm_content(mocker: MockerFixture) -> None: """random-mm image_url parts keep an explicit image_reference type.""" import aiohttp @@ -1655,6 +1668,38 @@ def tracking_add_field(self, name, value=None, **kwargs): assert json.loads(payload) == {"video_url": "https://example.com/ref.mp4"} +def test_image_and_inline_video_use_combined_reference_fields(mocker: MockerFixture) -> None: + """Image plus data:video must be image_reference + video_reference, not input_references.""" + import aiohttp + + content = [ + {"type": "image_url", "image_url": {"url": "https://example.com/ref.png"}}, + {"type": "video_url", "video_url": {"url": "data:video/mp4;base64,AAAA"}}, + ] + captured: list[tuple[str, object]] = [] + real_add_field = aiohttp.FormData.add_field + + def tracking_add_field(self, name, value=None, **kwargs): + captured.append((str(name), value)) + return real_add_field(self, name, value, **kwargs) + + mocker.patch.object(aiohttp.FormData, "add_field", tracking_add_field) + form = aiohttp.FormData() + _add_combined_video_form_references(form, content) + + field_names = [name for name, _ in captured] + assert field_names.count("image_reference") == 1 + assert field_names.count("video_reference") == 1 + assert "input_references" not in field_names + assert "input_reference" not in field_names + image_payload = next(value for name, value in captured if name == "image_reference") + video_payload = next(value for name, value in captured if name == "video_reference") + assert isinstance(image_payload, (str, bytes, bytearray)) + assert isinstance(video_payload, (str, bytes, bytearray)) + assert json.loads(image_payload) == {"image_url": "https://example.com/ref.png"} + assert json.loads(video_payload) == {"video_url": "data:video/mp4;base64,AAAA"} + + def test_video_unsupported_image_reference_raises() -> None: import aiohttp diff --git a/vllm_omni/benchmarks/patch/patch.py b/vllm_omni/benchmarks/patch/patch.py index fa32c59ada0..3d0cbddef2a 100644 --- a/vllm_omni/benchmarks/patch/patch.py +++ b/vllm_omni/benchmarks/patch/patch.py @@ -1312,12 +1312,9 @@ def _video_frames_from_payload(data: Mapping[str, object], request_body: Mapping def _is_structured_image_reference(reference: Mapping[str, object]) -> bool: - """True for API image_reference objects ({image_url}/{file_id}).""" + """True for API image_reference objects ({"image_url": "..."}).""" image_url = reference.get("image_url") - file_id = reference.get("file_id") - has_url = isinstance(image_url, str) and bool(image_url) - has_file_id = isinstance(file_id, str) and bool(file_id) - return has_url or has_file_id + return isinstance(image_url, str) and bool(image_url) def _is_structured_video_reference(reference: Mapping[str, object]) -> bool: @@ -1326,7 +1323,18 @@ def _is_structured_video_reference(reference: Mapping[str, object]) -> bool: return isinstance(video_url, str) and bool(video_url) -def _add_video_reference_to_form(form: aiohttp.FormData, reference: object) -> bool: +def _add_video_reference_to_form( + form: aiohttp.FormData, + reference: object, + *, + upload_inline_video: bool = True, +) -> bool: + candidates = reference if isinstance(reference, list) else [reference] + for item in candidates: + if isinstance(item, Mapping): + file_id = item.get("file_id") + if isinstance(file_id, str) and file_id: + raise ValueError("file_id is not supported yet") if isinstance(reference, dict) and "bytes" in reference: form.add_field( "input_reference", @@ -1343,7 +1351,10 @@ def _add_video_reference_to_form(form: aiohttp.FormData, reference: object) -> b if isinstance(reference, Mapping) and _is_structured_video_reference(reference): video_url = reference.get("video_url") # Inline data URLs are too large for a text form field (1MB part limit). - if isinstance(video_url, str) and video_url.startswith("data:video"): + # Skip the upload when an image_reference is also present: ``input_references`` + # cannot be combined with ``image_reference`` (HTTP 400). The server accepts + # ``image_reference`` together with ``video_reference``. + if upload_inline_video and isinstance(video_url, str) and video_url.startswith("data:video"): return _add_video_reference_to_form(form, video_url) form.add_field("video_reference", json.dumps(dict(reference))) return True @@ -1355,10 +1366,7 @@ def _add_video_reference_to_form(form: aiohttp.FormData, reference: object) -> b if reference and all(isinstance(item, Mapping) and _is_structured_video_reference(item) for item in reference): form.add_field("video_reference", json.dumps([dict(item) for item in reference])) return True - raise ValueError( - "Unsupported image_reference list; expected non-empty list of " - '{"image_url": "..."} and/or {"file_id": "..."} objects.' - ) + raise ValueError('Unsupported image_reference list; expected non-empty list of {"image_url": "..."} objects.') if isinstance(reference, str): if reference.startswith("data:video"): @@ -1400,11 +1408,37 @@ def _add_video_reference_to_form(form: aiohttp.FormData, reference: object) -> b raise ValueError( "Unsupported image_reference; expected upload bytes, local path/URL string, " - 'or {"image_url": "..."} / {"file_id": "..."} object ' + 'or {"image_url": "..."} object ' f"(got {type(reference).__name__})." ) +def _add_combined_video_form_references( + form: aiohttp.FormData, + multi_modal_content: Any, + extra_body: Mapping[str, Any] | None = None, +) -> None: + """Serialize image and video refs using a server-accepted field pair. + + ``image_reference`` may be combined with ``video_reference``. ``input_references`` + must be sent alone, so an inline ``data:video`` is uploaded only when no image + reference is present. + """ + extra_body = extra_body or {} + image_refs = list(_iter_image_reference_inputs(multi_modal_content)) + video_refs = list(_iter_video_reference_inputs(multi_modal_content)) + if not image_refs and extra_body.get("image_reference") is not None: + image_refs = [extra_body["image_reference"]] + if not video_refs and extra_body.get("video_reference") is not None: + video_refs = [extra_body["video_reference"]] + + upload_inline_video = not image_refs + if image_refs: + _add_video_reference_to_form(form, image_refs[0]) + if video_refs: + _add_video_reference_to_form(form, video_refs[0], upload_inline_video=upload_inline_video) + + def _add_video_extra_body_to_form( form: aiohttp.FormData, extra_body: Mapping[str, object], @@ -1960,25 +1994,7 @@ async def async_request_openai_videos_omni( form.add_field("size", str(request_body["size"])) _add_video_extra_body_to_form(form, extra_body, request_body) - image_reference_added = False - for reference in _iter_image_reference_inputs(request_func_input.multi_modal_content): - if _add_video_reference_to_form(form, reference): - image_reference_added = True - break - if not image_reference_added: - image_reference = extra_body.get("image_reference") - if image_reference is not None: - _add_video_reference_to_form(form, image_reference) - - video_reference_added = False - for reference in _iter_video_reference_inputs(request_func_input.multi_modal_content): - if _add_video_reference_to_form(form, reference): - video_reference_added = True - break - if not video_reference_added: - video_reference = extra_body.get("video_reference") - if video_reference is not None: - _add_video_reference_to_form(form, video_reference) + _add_combined_video_form_references(form, request_func_input.multi_modal_content, extra_body) headers = { "Authorization": f"Bearer {os.environ.get('OPENAI_API_KEY')}", From fdaef008f58bbba59c7470ce8f0b0cb0e13ebca1 Mon Sep 17 00:00:00 2001 From: wangyu <410167048@qq.com> Date: Mon, 21 Sep 2026 22:03:53 +0800 Subject: [PATCH 15/15] Enhance video and image reference handling in benchmarks - Refactored functions to classify and process image and video references, ensuring proper handling of bare URLs and data URIs. - Introduced new tests to validate the correct processing of video references, including oversized data videos and their handling in forms. - Improved error handling for unsupported video formats and added checks for reference types. These changes improve the robustness and clarity of the benchmarking framework's media reference handling. Signed-off-by: wangyu <410167048@qq.com> --- tests/benchmarks/patch/test_patch.py | 183 ++++++++++++++++++++- vllm_omni/benchmarks/patch/patch.py | 233 +++++++++++++++++++++------ 2 files changed, 361 insertions(+), 55 deletions(-) diff --git a/tests/benchmarks/patch/test_patch.py b/tests/benchmarks/patch/test_patch.py index 37410b65af0..b203fce59d2 100644 --- a/tests/benchmarks/patch/test_patch.py +++ b/tests/benchmarks/patch/test_patch.py @@ -1606,7 +1606,7 @@ def tracking_add_field(self, name, value=None, **kwargs): def test_video_reference_urls_from_random_mm_content(mocker: MockerFixture) -> None: - """random-mm data:video_url parts upload via input_references.""" + """random-mm data:video_url parts stay on video_reference.""" import aiohttp content = [ @@ -1628,11 +1628,15 @@ def tracking_add_field(self, name, value=None, **kwargs): mocker.patch.object(aiohttp.FormData, "add_field", tracking_add_field) form = aiohttp.FormData() assert _add_video_reference_to_form(form, refs[0]) is True - uploaded = next(value for name, value in captured if name == "input_references") - assert uploaded == base64.b64decode("AAAA") - assert "input_reference" not in [name for name, _ in captured] - assert "video_reference" not in [name for name, _ in captured] - assert "image_reference" not in [name for name, _ in captured] + + field_names = [name for name, _ in captured] + assert field_names.count("video_reference") == 1 + assert "input_references" not in field_names + assert "input_reference" not in field_names + assert "image_reference" not in field_names + payload = next(value for name, value in captured if name == "video_reference") + assert isinstance(payload, (str, bytes, bytearray)) + assert json.loads(payload) == {"video_url": "data:video/mp4;base64,AAAA"} def test_video_reference_https_url_from_random_mm_content(mocker: MockerFixture) -> None: @@ -1668,6 +1672,55 @@ def tracking_add_field(self, name, value=None, **kwargs): assert json.loads(payload) == {"video_url": "https://example.com/ref.mp4"} +@pytest.mark.parametrize( + ("reference", "field_name", "expected"), + [ + ("data:image/png;base64,AAAA", "image_reference", {"image_url": "data:image/png;base64,AAAA"}), + ("data:video/mp4;base64,AAAA", "video_reference", {"video_url": "data:video/mp4;base64,AAAA"}), + ("https://example.com/ref.png", "image_reference", {"image_url": "https://example.com/ref.png"}), + ("https://example.com/ref.mp4", "video_reference", {"video_url": "https://example.com/ref.mp4"}), + ], +) +def test_bare_reference_string_uses_matching_json_field( + reference: str, + field_name: str, + expected: dict[str, str], + mocker: MockerFixture, +) -> None: + """Image and video URLs use matching JSON fields; neither is a file upload.""" + import aiohttp + + captured: list[tuple[str, object]] = [] + real_add_field = aiohttp.FormData.add_field + + def tracking_add_field(self, name, value=None, **kwargs): + captured.append((str(name), value)) + return real_add_field(self, name, value, **kwargs) + + mocker.patch.object(aiohttp.FormData, "add_field", tracking_add_field) + assert _add_video_reference_to_form(aiohttp.FormData(), reference) is True + assert [name for name, _ in captured] == [field_name] + payload = captured[0][1] + assert isinstance(payload, (str, bytes, bytearray)) + assert json.loads(payload) == expected + + +def test_bare_video_string_is_not_collected_as_image() -> None: + content = ["https://example.com/ref.mp4", "data:video/mp4;base64,AAAA"] + assert list(_iter_image_reference_inputs(content)) == [] + assert list(_iter_video_reference_inputs(content)) == [ + {"video_url": "https://example.com/ref.mp4"}, + {"video_url": "data:video/mp4;base64,AAAA"}, + ] + + +def test_bare_http_without_media_extension_is_rejected() -> None: + import aiohttp + + with pytest.raises(ValueError, match="image or video extension"): + _add_video_reference_to_form(aiohttp.FormData(), "https://example.com/ref") + + def test_image_and_inline_video_use_combined_reference_fields(mocker: MockerFixture) -> None: """Image plus data:video must be image_reference + video_reference, not input_references.""" import aiohttp @@ -1700,13 +1753,127 @@ def tracking_add_field(self, name, value=None, **kwargs): assert json.loads(video_payload) == {"video_url": "data:video/mp4;base64,AAAA"} +def test_image_url_and_bare_data_video_string_use_json_fields(mocker: MockerFixture) -> None: + """Image URL plus a bare data:video string must not upload input_references.""" + import aiohttp + + content = [{"type": "image_url", "image_url": {"url": "https://example.com/ref.png"}}] + extra_body = {"video_reference": "data:video/mp4;base64,AAAA"} + captured: list[tuple[str, object]] = [] + real_add_field = aiohttp.FormData.add_field + + def tracking_add_field(self, name, value=None, **kwargs): + captured.append((str(name), value)) + return real_add_field(self, name, value, **kwargs) + + mocker.patch.object(aiohttp.FormData, "add_field", tracking_add_field) + form = aiohttp.FormData() + _add_combined_video_form_references(form, content, extra_body) + + field_names = [name for name, _ in captured] + assert field_names.count("image_reference") == 1 + assert field_names.count("video_reference") == 1 + assert "input_references" not in field_names + assert "input_reference" not in field_names + image_payload = next(value for name, value in captured if name == "image_reference") + video_payload = next(value for name, value in captured if name == "video_reference") + assert isinstance(image_payload, (str, bytes, bytearray)) + assert isinstance(video_payload, (str, bytes, bytearray)) + assert json.loads(image_payload) == {"image_url": "https://example.com/ref.png"} + assert json.loads(video_payload) == {"video_url": "data:video/mp4;base64,AAAA"} + + captured.clear() + assert _add_video_reference_to_form(form, "data:video/mp4;base64,AAAA") is True + assert [name for name, _ in captured] == ["video_reference"] + + +def _oversized_data_video_url() -> str: + """data:video whose JSON text part is larger than 1MB.""" + return "data:video/mp4;base64," + ("A" * (1024 * 1024)) + + +def test_oversized_data_video_uploads_as_input_references(mocker: MockerFixture) -> None: + """A lone data:video over the 1MB text limit is uploaded, not sent as JSON.""" + import aiohttp + + video_url = _oversized_data_video_url() + captured: list[tuple[str, object]] = [] + real_add_field = aiohttp.FormData.add_field + + def tracking_add_field(self, name, value=None, **kwargs): + captured.append((str(name), value)) + return real_add_field(self, name, value, **kwargs) + + mocker.patch.object(aiohttp.FormData, "add_field", tracking_add_field) + assert _add_video_reference_to_form(aiohttp.FormData(), {"video_url": video_url}) is True + uploaded = next(value for name, value in captured if name == "input_references") + assert uploaded == base64.b64decode("A" * (1024 * 1024)) + field_names = [name for name, _ in captured] + assert "video_reference" not in field_names + assert "image_reference" not in field_names + + +def test_oversized_data_video_with_image_stays_json(mocker: MockerFixture) -> None: + """input_references cannot be combined with image_reference, even for a large video.""" + import aiohttp + + content = [{"type": "image_url", "image_url": {"url": "https://example.com/ref.png"}}] + extra_body = {"video_reference": _oversized_data_video_url()} + captured: list[tuple[str, object]] = [] + real_add_field = aiohttp.FormData.add_field + + def tracking_add_field(self, name, value=None, **kwargs): + captured.append((str(name), value)) + return real_add_field(self, name, value, **kwargs) + + mocker.patch.object(aiohttp.FormData, "add_field", tracking_add_field) + _add_combined_video_form_references(aiohttp.FormData(), content, extra_body) + field_names = [name for name, _ in captured] + assert "image_reference" in field_names + assert "video_reference" in field_names + assert "input_references" not in field_names + + +def test_image_upload_bytes_and_structured_video_use_json_fields(mocker: MockerFixture) -> None: + """Image upload bytes plus a video URL must not use singular input_reference.""" + import aiohttp + + image_bytes = base64.b64decode(_MIN_PNG_B64) + extra_body = { + "image_reference": {"bytes": image_bytes, "content_type": "image/png"}, + "video_reference": {"video_url": "https://example.com/ref.mp4"}, + } + captured: list[tuple[str, object]] = [] + real_add_field = aiohttp.FormData.add_field + + def tracking_add_field(self, name, value=None, **kwargs): + captured.append((str(name), value)) + return real_add_field(self, name, value, **kwargs) + + mocker.patch.object(aiohttp.FormData, "add_field", tracking_add_field) + form = aiohttp.FormData() + _add_combined_video_form_references(form, None, extra_body) + + field_names = [name for name, _ in captured] + assert field_names.count("image_reference") == 1 + assert field_names.count("video_reference") == 1 + assert "input_reference" not in field_names + assert "input_references" not in field_names + image_payload = next(value for name, value in captured if name == "image_reference") + video_payload = next(value for name, value in captured if name == "video_reference") + assert isinstance(image_payload, (str, bytes, bytearray)) + assert isinstance(video_payload, (str, bytes, bytearray)) + assert json.loads(image_payload) == {"image_url": f"data:image/png;base64,{_MIN_PNG_B64}"} + assert json.loads(video_payload) == {"video_url": "https://example.com/ref.mp4"} + + def test_video_unsupported_image_reference_raises() -> None: import aiohttp form = aiohttp.FormData() - with pytest.raises(ValueError, match="Unsupported image_reference"): + with pytest.raises(ValueError, match="Unsupported reference"): _add_video_reference_to_form(form, {"not_a_supported_key": "x"}) - with pytest.raises(ValueError, match="Unsupported image_reference"): + with pytest.raises(ValueError, match="Unsupported reference"): _add_video_reference_to_form(form, "/tmp/does-not-exist-ref.png") diff --git a/vllm_omni/benchmarks/patch/patch.py b/vllm_omni/benchmarks/patch/patch.py index af3dea261dc..cfc609c3e34 100644 --- a/vllm_omni/benchmarks/patch/patch.py +++ b/vllm_omni/benchmarks/patch/patch.py @@ -19,6 +19,7 @@ from datetime import datetime from pathlib import Path from typing import TYPE_CHECKING, Any, Literal +from urllib.parse import urlparse import aiohttp import numpy as np @@ -906,13 +907,41 @@ def _guess_mime_type(path: str) -> str: return mime or "application/octet-stream" +_IMAGE_REFERENCE_SUFFIXES = frozenset({".png", ".jpg", ".jpeg", ".webp", ".gif", ".bmp", ".heic", ".heif"}) +_VIDEO_REFERENCE_SUFFIXES = frozenset({".mp4", ".mov", ".webm", ".mkv", ".m4v"}) + + +def _string_reference_kind(reference: str) -> str | None: + """Classify a bare reference string as image, video, or file. + + ``data:image`` / ``data:video`` carry their type in the URL. Bare http(s) + URLs use the path extension. An existing local path is a file upload, not + an image or a video URL. + """ + if reference.startswith("data:image"): + return "image" + if reference.startswith("data:video"): + return "video" + if reference.startswith(("http://", "https://")): + suffix = Path(urlparse(reference).path).suffix.lower() + if suffix in _VIDEO_REFERENCE_SUFFIXES: + return "video" + if suffix in _IMAGE_REFERENCE_SUFFIXES: + return "image" + return None + local_path = reference.removeprefix("file://") + if local_path and os.path.exists(local_path): + return "file" + return None + + def _iter_image_reference_inputs(value: Any) -> Iterable[Any]: """Yield image references from benchmark multimodal content. ``random-mm`` image buckets arrive as OpenAI chat parts ``{"type": "image_url", "image_url": {"url": ...}}``. Yield - ``{"image_url": url}`` so the form helper keeps an explicit image - type (symmetric with ``_iter_video_reference_inputs``). + ``{"image_url": url}`` so the form helper keeps an explicit image type. + Bare video strings are left for ``_iter_video_reference_inputs``. """ if value is None: return @@ -921,6 +950,8 @@ def _iter_image_reference_inputs(value: Any) -> Iterable[Any]: yield from _iter_image_reference_inputs(item) return if not isinstance(value, dict): + if isinstance(value, str) and _string_reference_kind(value) == "video": + return yield value return @@ -944,10 +975,9 @@ def _iter_video_reference_inputs(value: Any) -> Iterable[dict[str, str]]: """Yield structured video references from benchmark multimodal content. ``random-mm`` video buckets arrive as OpenAI chat parts - ``{"type": "video_url", "video_url": {"url": ...}}``. Yield - ``{"video_url": url}`` so ``_add_video_reference_to_form`` keeps the - video branch (HTTP(S) bare strings would otherwise become - ``image_reference``). + ``{"type": "video_url", "video_url": {"url": ...}}``, or as a bare + ``data:video`` / video http(s) string. Yield ``{"video_url": url}`` so the + form helper keeps the video branch. """ if value is None: return @@ -956,6 +986,8 @@ def _iter_video_reference_inputs(value: Any) -> Iterable[dict[str, str]]: yield from _iter_video_reference_inputs(item) return if not isinstance(value, dict): + if isinstance(value, str) and _string_reference_kind(value) == "video": + yield {"video_url": value} return if value.get("type") == "video_url": @@ -1444,12 +1476,112 @@ def _is_structured_video_reference(reference: Mapping[str, object]) -> bool: return isinstance(video_url, str) and bool(video_url) +_VIDEO_REFERENCE_JSON_MAX_BYTES = 1024 * 1024 + + +def _data_video_json_exceeds_text_limit(video_url: str) -> bool: + """True when a data:video URL would exceed the ~1MB multipart text-part limit.""" + if not video_url.startswith("data:video"): + return False + encoded = json.dumps({"video_url": video_url}).encode("utf-8") + return len(encoded) > _VIDEO_REFERENCE_JSON_MAX_BYTES + + +def _add_data_video_upload(form: aiohttp.FormData, video_url: str) -> bool: + """Upload one inline video as ``input_references`` instead of a JSON text part.""" + header, _, payload = video_url.partition(",") + if not payload: + raise ValueError(f"Unsupported video data URL: {video_url[:64]!r}") + try: + video_bytes = base64.b64decode(payload) + except (ValueError, TypeError) as exc: + raise ValueError("video data URL is not valid base64") from exc + mime = header[len("data:") :].split(";", 1)[0] or "video/mp4" + suffix = ".mp4" if mime.endswith("mp4") else ".bin" + form.add_field( + "input_references", + video_bytes, + filename=f"benchmark-reference{suffix}", + content_type=mime, + ) + return True + + +def _file_bytes_as_data_url(raw: bytes, mime: str) -> str: + encoded = base64.b64encode(raw).decode("ascii") + return f"data:{mime};base64,{encoded}" + + +def _image_reference_json_value(reference: object) -> object: + """Turn an image file into a JSON ``image_reference`` when it must share a form. + + ``input_reference`` cannot be combined with ``video_reference``. Image URLs + stay URLs. Upload bytes and local image files become ``data:image`` URLs. + """ + if isinstance(reference, Mapping) and "bytes" in reference and not _is_structured_image_reference(reference): + raw = reference["bytes"] + if not isinstance(raw, (bytes, bytearray)): + raise ValueError(f"image reference bytes must be bytes (got {type(raw).__name__}).") + content_type = reference.get("content_type", "image/png") + if not isinstance(content_type, str) or not content_type.startswith("image/"): + content_type = "image/png" + return {"image_url": _file_bytes_as_data_url(bytes(raw), content_type.split(";", 1)[0])} + if isinstance(reference, str): + kind = _string_reference_kind(reference) + if kind == "image": + return {"image_url": reference} + if kind == "file": + local_path = reference.removeprefix("file://") + mime = _guess_mime_type(local_path) + if not mime.startswith("image/"): + mime = "image/png" + with open(local_path, "rb") as handle: + return {"image_url": _file_bytes_as_data_url(handle.read(), mime)} + return reference + + +def _video_reference_json_value(reference: object) -> object: + """Turn a video file into a JSON ``video_reference`` when it must share a form. + + ``input_reference`` cannot be combined with ``image_reference``. Video URLs + stay URLs. Local video files become ``data:video`` URLs. + """ + if isinstance(reference, Mapping) and "bytes" in reference and not _is_structured_video_reference(reference): + raw = reference["bytes"] + if not isinstance(raw, (bytes, bytearray)): + raise ValueError(f"video reference bytes must be bytes (got {type(raw).__name__}).") + content_type = reference.get("content_type", "video/mp4") + if not isinstance(content_type, str) or not content_type.startswith("video/"): + content_type = "video/mp4" + return {"video_url": _file_bytes_as_data_url(bytes(raw), content_type.split(";", 1)[0])} + if isinstance(reference, str): + kind = _string_reference_kind(reference) + if kind == "video": + return {"video_url": reference} + if kind == "file": + local_path = reference.removeprefix("file://") + mime = _guess_mime_type(local_path) + if not mime.startswith("video/"): + mime = "video/mp4" + with open(local_path, "rb") as handle: + return {"video_url": _file_bytes_as_data_url(handle.read(), mime)} + return reference + + def _add_video_reference_to_form( form: aiohttp.FormData, reference: object, *, upload_inline_video: bool = True, ) -> bool: + """Encode one reference: image URL, video URL, or file upload. + + Image URLs use ``image_reference``. Video URLs use ``video_reference``. + A lone ``data:video`` whose JSON text exceeds 1MB is uploaded as + ``input_references``. Local paths and raw bytes use ``input_reference``. + ``upload_inline_video`` must be false when an image is on the same form: + ``input_references`` cannot be combined with ``image_reference``. + """ candidates = reference if isinstance(reference, list) else [reference] for item in candidates: if isinstance(item, Mapping): @@ -1471,12 +1603,8 @@ def _add_video_reference_to_form( if isinstance(reference, Mapping) and _is_structured_video_reference(reference): video_url = reference.get("video_url") - # Inline data URLs are too large for a text form field (1MB part limit). - # Skip the upload when an image_reference is also present: ``input_references`` - # cannot be combined with ``image_reference`` (HTTP 400). The server accepts - # ``image_reference`` together with ``video_reference``. - if upload_inline_video and isinstance(video_url, str) and video_url.startswith("data:video"): - return _add_video_reference_to_form(form, video_url) + if upload_inline_video and isinstance(video_url, str) and _data_video_json_exceeds_text_limit(video_url): + return _add_data_video_upload(form, video_url) form.add_field("video_reference", json.dumps(dict(reference))) return True @@ -1487,37 +1615,25 @@ def _add_video_reference_to_form( if reference and all(isinstance(item, Mapping) and _is_structured_video_reference(item) for item in reference): form.add_field("video_reference", json.dumps([dict(item) for item in reference])) return True - raise ValueError('Unsupported image_reference list; expected non-empty list of {"image_url": "..."} objects.') + raise ValueError( + "Unsupported reference list; expected non-empty list of " + '{"image_url": "..."} or {"video_url": "..."} objects.' + ) if isinstance(reference, str): - if reference.startswith("data:video"): - header, _, payload = reference.partition(",") - if not payload: - raise ValueError(f"Unsupported video data URL: {reference[:64]!r}") - try: - video_bytes = base64.b64decode(payload) - except (ValueError, TypeError) as exc: - raise ValueError("video data URL is not valid base64") from exc - mime = header[len("data:") :].split(";", 1)[0] or "video/mp4" - suffix = ".mp4" if mime.endswith("mp4") else ".bin" - form.add_field( - # Plural field persists the container to disk. Singular - # ``input_reference`` would decode every frame in the API - # process and trip Starlette / MiniMax size limits on - # random-mm videos. - "input_references", - video_bytes, - filename=f"benchmark-reference{suffix}", - content_type=mime, - ) - return True - if reference.startswith(("data:image", "http://", "https://")): + kind = _string_reference_kind(reference) + if kind == "image": form.add_field("image_reference", json.dumps({"image_url": reference})) return True - local_path = reference.removeprefix("file://") - if os.path.exists(local_path): - with open(local_path, "rb") as f: - reference_bytes = f.read() + if kind == "video": + if upload_inline_video and _data_video_json_exceeds_text_limit(reference): + return _add_data_video_upload(form, reference) + form.add_field("video_reference", json.dumps({"video_url": reference})) + return True + if kind == "file": + local_path = reference.removeprefix("file://") + with open(local_path, "rb") as handle: + reference_bytes = handle.read() form.add_field( "input_reference", reference_bytes, @@ -1525,11 +1641,16 @@ def _add_video_reference_to_form( content_type=_guess_mime_type(local_path), ) return True - raise ValueError(f"Unsupported image_reference path or URL: {reference!r}") + if reference.startswith(("http://", "https://")): + raise ValueError( + "Bare http(s) reference needs an image or video extension " + f"({', '.join(sorted(_IMAGE_REFERENCE_SUFFIXES | _VIDEO_REFERENCE_SUFFIXES))}); " + f"got {reference!r}." + ) + raise ValueError(f"Unsupported reference path or URL: {reference!r}") raise ValueError( - "Unsupported image_reference; expected upload bytes, local path/URL string, " - 'or {"image_url": "..."} object ' + "Unsupported reference; expected image URL, video URL, upload bytes, or a local file " f"(got {type(reference).__name__})." ) @@ -1541,9 +1662,12 @@ def _add_combined_video_form_references( ) -> None: """Serialize image and video refs using a server-accepted field pair. - ``image_reference`` may be combined with ``video_reference``. ``input_references`` - must be sent alone, so an inline ``data:video`` is uploaded only when no image - reference is present. + Alone, each reference uses its own field: image URL → ``image_reference``, + video URL → ``video_reference``, file → ``input_reference``. A lone inline + video whose JSON text exceeds 1MB is uploaded as ``input_references``. + Together, that upload cannot be combined with ``image_reference``, so both + sides stay on the JSON fields. A file paired with the other media is rewritten + as a data URL of the matching type. """ extra_body = extra_body or {} image_refs = list(_iter_image_reference_inputs(multi_modal_content)) @@ -1553,11 +1677,26 @@ def _add_combined_video_form_references( if not video_refs and extra_body.get("video_reference") is not None: video_refs = [extra_body["video_reference"]] - upload_inline_video = not image_refs + if image_refs and video_refs: + for raw in (image_refs[0], video_refs[0]): + candidates = raw if isinstance(raw, list) else [raw] + for item in candidates: + if isinstance(item, Mapping): + file_id = item.get("file_id") + if isinstance(file_id, str) and file_id: + raise ValueError("file_id is not supported yet") + _add_video_reference_to_form(form, _image_reference_json_value(image_refs[0])) + _add_video_reference_to_form( + form, + _video_reference_json_value(video_refs[0]), + upload_inline_video=False, + ) + return + if image_refs: _add_video_reference_to_form(form, image_refs[0]) if video_refs: - _add_video_reference_to_form(form, video_refs[0], upload_inline_video=upload_inline_video) + _add_video_reference_to_form(form, video_refs[0]) def _add_video_extra_body_to_form(