From 39899bd5b7e25a1a7db5a36e8e9bfbee3dff4364 Mon Sep 17 00:00:00 2001 From: natureofnature Date: Tue, 25 Aug 2026 12:23:25 -0400 Subject: [PATCH 1/6] [CI/Build] Add OmniInteract nightly E2E Signed-off-by: natureofnature Co-authored-by: Ruirui Yang | Rein <73573651+R2-Y@users.noreply.github.com> --- .buildkite/cuda/test-nightly.yml | 10 +- docs/cli/bench/serve.md | 6 + .../test_minicpmo_4_5_duplex_expansion.py | 120 ++++++++++++++++++ 3 files changed, 134 insertions(+), 2 deletions(-) diff --git a/.buildkite/cuda/test-nightly.yml b/.buildkite/cuda/test-nightly.yml index 9f44ba2b0ce..96eb48eb190 100644 --- a/.buildkite/cuda/test-nightly.yml +++ b/.buildkite/cuda/test-nightly.yml @@ -25,9 +25,15 @@ steps: mirror_hardwares: h100_2 - label: ":full_moon: Omni · MiniCPM-o 4.5 Duplex Test" - timeout_in_minutes: 50 + timeout_in_minutes: 120 commands: - - pytest -s -v tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py -m "full_model and cuda and H100 and omni and cards_1" --run-level "full_model" + - export VLLM_OMNI_RUN_OMNIINTERACT_E2E=1 + - | + set +e + pytest -s -v tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py -m "full_model and cuda and H100 and omni and cards_1" --run-level "full_model" --basetemp=tests/e2e/online_serving/omniinteract-nightly + EXIT=$$? + buildkite-agent artifact upload "tests/e2e/online_serving/omniinteract-nightly/**/*" || EXIT=1 + exit $$EXIT mirror_hardwares: h100_1 - label: ":full_moon: Omni · Doc Test with L4" diff --git a/docs/cli/bench/serve.md b/docs/cli/bench/serve.md index 3e16ae5542f..7c61bd6336b 100644 --- a/docs/cli/bench/serve.md +++ b/docs/cli/bench/serve.md @@ -307,6 +307,12 @@ omitted from the official manifest; `audio_clipped_bytes` records output beyond TTFT, TTFP, and RTF start at client receipt of `response.created`. TPOT/ITL use engine stage-0 timing; ITL is emitted only when every token interval is present. +The CUDA Nightly functional gate runs four deterministic cases from each OmniInteract subset (12 videos total), with no +warmups and a maximum concurrency of two. It requires every case to commit its input, complete any emitted response +lifecycles, and publish parseable WAV, transcript, event, and result artifacts. A valid LISTEN-only case may have no response +audio or transcript chunks. Official-manifest eligibility is reported separately because clipped or cancelled output is a +benchmark-quality signal, not a transport failure. This gate does not score answer accuracy. + ### Multi-Modal Benchmark
diff --git a/tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py b/tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py index 5b35c4b53fe..60ba59c88a6 100644 --- a/tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py +++ b/tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py @@ -7,11 +7,17 @@ import asyncio import json +import os +import wave from pathlib import Path from types import SimpleNamespace import pytest +from tests.e2e.accuracy.qwen3_omni.qwen3_omni_acc_bench_core import ( + find_vllm_cli, + run_vllm_bench_subprocess, +) from tests.e2e.online_serving.helpers.minicpmo_4_5_duplex import ( SERVER_PARAMS, SOFT_INTERRUPT_SHA256, @@ -30,9 +36,12 @@ run_soft_interrupt, ) from tests.helpers.mark import hardware_test +from vllm_omni.benchmarks.data_modules.omniinteract_dataset import DEFAULT_OMNIINTERACT_REPO pytestmark = [pytest.mark.full_model, pytest.mark.omni] +OMNIINTERACT_NIGHTLY_DATASET = f"{DEFAULT_OMNIINTERACT_REPO}@e195f75fe2666fcc5fe74f537ae49ca143a79969" + @hardware_test(res={"cuda": "H100", "npu": "A3"}, num_cards=1) @pytest.mark.parametrize("omni_server", SERVER_PARAMS, indirect=True) @@ -112,3 +121,114 @@ def test_duplex_server_vad_hard_interrupt(omni_server) -> None: ) ) assert result["ok"] is True, json.dumps(result, ensure_ascii=False, indent=2) + + +@hardware_test(res={"cuda": "H100"}, num_cards=1) +@pytest.mark.skipif( + os.environ.get("VLLM_OMNI_RUN_OMNIINTERACT_E2E") != "1", + reason="enable the real-time OmniInteract dataset E2E", +) +@pytest.mark.parametrize("subset", ("1q1a", "1q1a_math", "1qna")) +@pytest.mark.parametrize("omni_server", SERVER_PARAMS, indirect=True) +def test_minicpmo_4_5_omniinteract_nightly( + omni_server, + tmp_path: Path, + subset: str, +) -> None: + dataset_path = os.environ.get("OMNIINTERACT_ROOT", "").strip() or OMNIINTERACT_NIGHTLY_DATASET + run_root = tmp_path / subset + artifact_root = run_root / "artifacts" + result_filename = "benchmark_result.json" + run_vllm_bench_subprocess( + find_vllm_cli(), + [ + "bench", + "serve", + "--omni", + "--trust-remote-code", + "--host", + omni_server.host, + "--port", + str(omni_server.port), + "--backend", + "openai-realtime-duplex", + "--endpoint", + "/v1/realtime", + "--model", + omni_server.model, + "--dataset-name", + "omniinteract", + "--dataset-path", + dataset_path, + "--omniinteract-subsets", + subset, + "--omniinteract-ref-audio", + str(resolve_ref_audio()), + "--omniinteract-output-dir", + str(artifact_root), + "--omniinteract-max-video-duration-s", + "600", + "--num-warmups", + "0", + "--num-prompts", + "4", + "--max-concurrency", + "2", + "--request-rate", + "inf", + "--disable-shuffle", + "--save-result", + "--result-dir", + str(run_root), + "--result-filename", + result_filename, + ], + ) + + benchmark_result = json.loads((run_root / result_filename).read_text()) + artifact_summary = benchmark_result["omniinteract"] + assert artifact_summary["artifacts_complete"] is True + assert (artifact_summary["total"], artifact_summary["success"], artifact_summary["failed"]) == (4, 4, 0) + + batch_summary = json.loads((artifact_root / "batch_summary.json").read_text()) + assert (batch_summary["total"], batch_summary["success"], batch_summary["failed"]) == (4, 4, 0) + manifest_rows = [ + json.loads(row) for row in (artifact_root / "official_eval_manifest.jsonl").read_text().splitlines() + ] + assert len(manifest_rows) == batch_summary["eligible_for_official_eval"] + assert {row["video"] for row in manifest_rows} == { + result["video"] for result in batch_summary["results"] if result["eligible_for_official_eval"] + } + assert {result["subset"] for result in batch_summary["results"]} == {subset} + assert len({result["video"] for result in batch_summary["results"]}) == 4 + + for result in batch_summary["results"]: + assert result["success"] is True + assert result["input_audio_chunks"] > 0 + assert result["input_video_frames"] > 0 + sample_root = Path(result["output_dir"]) + assert all( + (sample_root / name).is_file() + for name in (".done", "output.wav", "wav_transcript.json", "events.json", "result.json") + ) + with wave.open(str(sample_root / "output.wav"), "rb") as output_wav: + assert output_wav.getnframes() > 0 + assert ( + output_wav.getframerate(), + output_wav.getnchannels(), + output_wav.getsampwidth(), + output_wav.getcomptype(), + ) == (24_000, 1, 2, "NONE") + transcript = json.loads((sample_root / "wav_transcript.json").read_text()) + assert isinstance(transcript["chunks"], list) + assert transcript["timestamp_semantics"] + published_result = json.loads((sample_root / "result.json").read_text()) + assert published_result["video"] == result["video"] + assert published_result["success"] is True + events = json.loads((sample_root / "events.json").read_text()) + created = {event["response"]["id"] for event in events if event.get("type") == "response.created"} + done = {event["response"]["id"] for event in events if event.get("type") == "response.done"} + assert created == done + assert len(created) == result["responses"] + if not result["eligible_for_official_eval"]: + assert result["official_eval_ineligible_reasons"] From a1be76353e22b943f76924a72ba1e6fc9dad7409 Mon Sep 17 00:00:00 2001 From: natureofnature Date: Mon, 31 Aug 2026 04:35:49 -0400 Subject: [PATCH 2/6] [CI] Allow cold OmniInteract dataset bootstrap Signed-off-by: natureofnature --- .buildkite/cuda/test-nightly.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.buildkite/cuda/test-nightly.yml b/.buildkite/cuda/test-nightly.yml index 96eb48eb190..46b8032f157 100644 --- a/.buildkite/cuda/test-nightly.yml +++ b/.buildkite/cuda/test-nightly.yml @@ -25,7 +25,7 @@ steps: mirror_hardwares: h100_2 - label: ":full_moon: Omni · MiniCPM-o 4.5 Duplex Test" - timeout_in_minutes: 120 + timeout_in_minutes: 300 commands: - export VLLM_OMNI_RUN_OMNIINTERACT_E2E=1 - | From a5cd8a5d78a8270a3ffe2f172307241613de6e24 Mon Sep 17 00:00:00 2001 From: natureofnature Date: Tue, 1 Sep 2026 05:46:10 -0400 Subject: [PATCH 3/6] [CI/Build] Isolate OmniInteract nightly from duplex checks Signed-off-by: natureofnature --- .buildkite/cuda/test-nightly.yml | 8 +++++++- docs/cli/bench/serve.md | 4 ++-- tests/benchmarks/test_omniinteract.py | 19 ++++++++++++------- tests/buildkite/test_upload_pipeline.py | 19 +++++++++++++++++++ .../test_minicpmo_4_5_duplex_expansion.py | 14 +++++++------- 5 files changed, 47 insertions(+), 17 deletions(-) diff --git a/.buildkite/cuda/test-nightly.yml b/.buildkite/cuda/test-nightly.yml index 46b8032f157..fa9d1311c2b 100644 --- a/.buildkite/cuda/test-nightly.yml +++ b/.buildkite/cuda/test-nightly.yml @@ -25,12 +25,18 @@ steps: mirror_hardwares: h100_2 - label: ":full_moon: Omni · MiniCPM-o 4.5 Duplex Test" + timeout_in_minutes: 50 + commands: + - pytest -s -v tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py -m "full_model and cuda and H100 and omni and cards_1" --run-level "full_model" + mirror_hardwares: h100_1 + + - label: ":full_moon: Omni · MiniCPM-o 4.5 OmniInteract Nightly" timeout_in_minutes: 300 commands: - export VLLM_OMNI_RUN_OMNIINTERACT_E2E=1 - | set +e - pytest -s -v tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py -m "full_model and cuda and H100 and omni and cards_1" --run-level "full_model" --basetemp=tests/e2e/online_serving/omniinteract-nightly + pytest -s -v tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py -k "omniinteract" -m "full_model and cuda and H100 and omni and cards_1" --run-level "full_model" --basetemp=tests/e2e/online_serving/omniinteract-nightly EXIT=$$? buildkite-agent artifact upload "tests/e2e/online_serving/omniinteract-nightly/**/*" || EXIT=1 exit $$EXIT diff --git a/docs/cli/bench/serve.md b/docs/cli/bench/serve.md index 7c61bd6336b..de1f21d3eb4 100644 --- a/docs/cli/bench/serve.md +++ b/docs/cli/bench/serve.md @@ -307,8 +307,8 @@ omitted from the official manifest; `audio_clipped_bytes` records output beyond TTFT, TTFP, and RTF start at client receipt of `response.created`. TPOT/ITL use engine stage-0 timing; ITL is emitted only when every token interval is present. -The CUDA Nightly functional gate runs four deterministic cases from each OmniInteract subset (12 videos total), with no -warmups and a maximum concurrency of two. It requires every case to commit its input, complete any emitted response +The dedicated CUDA Nightly benchmark gate runs four deterministic cases from each OmniInteract subset (12 videos total), +with no warmups and a maximum concurrency of two. It requires every case to commit its input, complete any emitted response lifecycles, and publish parseable WAV, transcript, event, and result artifacts. A valid LISTEN-only case may have no response audio or transcript chunks. Official-manifest eligibility is reported separately because clipped or cancelled output is a benchmark-quality signal, not a transport failure. This gate does not score answer accuracy. diff --git a/tests/benchmarks/test_omniinteract.py b/tests/benchmarks/test_omniinteract.py index c38e001bf48..70fc8168055 100644 --- a/tests/benchmarks/test_omniinteract.py +++ b/tests/benchmarks/test_omniinteract.py @@ -203,18 +203,19 @@ def test_archive_is_safe_and_atomically_shared(tmp_path: Path, monkeypatch: pyte assert not list(target.glob(".tmp-*")) -def test_hub_archive_uses_vllm_filesystem(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): +@pytest.mark.parametrize("dataset_repo", ["org/repo", "org/repo@revision"]) +def test_hub_archive_uses_vllm_filesystem(tmp_path: Path, monkeypatch: pytest.MonkeyPatch, dataset_repo: str): source = tmp_path / "source.tar" _archive(source, "data/1q1a/video_json_map.json") class FS: def get_file(self, remote: str, local: str) -> None: - assert remote == "datasets/org/repo/data.tar.gz" + assert remote == f"datasets/{dataset_repo}/data.tar.gz" Path(local).write_bytes(source.read_bytes()) monkeypatch.setenv("HF_HOME", str(tmp_path / "hf")) monkeypatch.setattr(data, "hf_fs", lambda: FS()) - assert (data.resolve_omniinteract_root(None, "org/repo") / "1q1a").is_dir() + assert (data.resolve_omniinteract_root(None, dataset_repo) / "1q1a").is_dir() def test_media_commands_are_bounded(monkeypatch: pytest.MonkeyPatch, tmp_path: Path): @@ -285,7 +286,8 @@ def test_response_ledger_rejects_identity_errors(events, match: str): class _CompletionClient: def __init__(self, collector: RealtimeEventCollector): - self.events, self.acks = collector, [] + self.events = collector + self.acks: list[tuple[str, int]] = [] def raise_if_reader_stopped(self) -> None: return None @@ -430,8 +432,11 @@ def test_atomic_write_preserves_destination_on_failure(tmp_path: Path, monkeypat class _RealtimeClient: instances: list[_RealtimeClient] = [] - def __init__(self, url: str, **kwargs): - self.url, self.events, self.acks, self.configure_kwargs = url, RealtimeEventCollector(), [], {} + def __init__(self, url: str, **kwargs: object): + self.url = url + self.events = RealtimeEventCollector() + self.acks: list[tuple[str, int]] = [] + self.configure_kwargs: dict[str, object] = {} self.instances.append(self) async def __aenter__(self): @@ -440,7 +445,7 @@ async def __aenter__(self): async def __aexit__(self, *args): return None - async def configure(self, model: str, **kwargs) -> None: + async def configure(self, model: str, **kwargs: object) -> None: self.configure_kwargs = kwargs self.events.add({"type": "session.created", "session": {"capabilities": {"chunk_period_ms": 1000}}}) diff --git a/tests/buildkite/test_upload_pipeline.py b/tests/buildkite/test_upload_pipeline.py index c994f8dd29f..c15194a8d77 100644 --- a/tests/buildkite/test_upload_pipeline.py +++ b/tests/buildkite/test_upload_pipeline.py @@ -110,6 +110,25 @@ def test_yaml_gated_l45_only_does_not_unconditionally_build_image() -> None: assert "key: upload-weekly-pipeline" in rendered +def test_minicpmo_omniinteract_nightly_isolated_from_duplex() -> None: + path = Path(".buildkite/cuda/test-nightly.yml") + rendered = _render_test_pipeline(yaml.safe_load(path.read_text()), changed_files=None) + steps = [step for group in rendered["steps"] for step in group.get("steps", [])] + by_label = {step.get("label"): step for step in steps} + + duplex = by_label[":full_moon: Omni · MiniCPM-o 4.5 Duplex Test"] + assert duplex["timeout_in_minutes"] == 50 + assert len(duplex["commands"]) == 1 + assert "VLLM_OMNI_RUN_OMNIINTERACT_E2E" not in duplex["commands"][0] + + omniinteract = by_label[":full_moon: Omni · MiniCPM-o 4.5 OmniInteract Nightly"] + assert omniinteract["timeout_in_minutes"] == 300 + commands = "\n".join(omniinteract["commands"]) + assert "VLLM_OMNI_RUN_OMNIINTERACT_E2E=1" in commands + assert '-k "omniinteract"' in commands + assert "artifact upload" in commands + + def test_yaml_gated_l2_still_enables_image_via_ready_base() -> None: rendered = _render([".buildkite/cuda/test-ready.yml"]) assert 'build.pull_request.labels includes "ready"' in rendered diff --git a/tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py b/tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py index 60ba59c88a6..edcbae099f5 100644 --- a/tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py +++ b/tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py @@ -8,16 +8,14 @@ import asyncio import json import os +import shutil +import subprocess import wave from pathlib import Path from types import SimpleNamespace import pytest -from tests.e2e.accuracy.qwen3_omni.qwen3_omni_acc_bench_core import ( - find_vllm_cli, - run_vllm_bench_subprocess, -) from tests.e2e.online_serving.helpers.minicpmo_4_5_duplex import ( SERVER_PARAMS, SOFT_INTERRUPT_SHA256, @@ -139,9 +137,11 @@ def test_minicpmo_4_5_omniinteract_nightly( run_root = tmp_path / subset artifact_root = run_root / "artifacts" result_filename = "benchmark_result.json" - run_vllm_bench_subprocess( - find_vllm_cli(), + vllm = shutil.which("vllm") + assert vllm is not None, "Could not find `vllm` on PATH" + subprocess.run( [ + vllm, "bench", "serve", "--omni", @@ -183,6 +183,7 @@ def test_minicpmo_4_5_omniinteract_nightly( "--result-filename", result_filename, ], + check=True, ) benchmark_result = json.loads((run_root / result_filename).read_text()) @@ -212,7 +213,6 @@ def test_minicpmo_4_5_omniinteract_nightly( for name in (".done", "output.wav", "wav_transcript.json", "events.json", "result.json") ) with wave.open(str(sample_root / "output.wav"), "rb") as output_wav: - assert output_wav.getnframes() > 0 assert ( output_wav.getframerate(), output_wav.getnchannels(), From 68b3dab4fab7f9996955f6627e388529ce6dd03e Mon Sep 17 00:00:00 2001 From: natureofnature Date: Tue, 1 Sep 2026 23:21:50 -0400 Subject: [PATCH 4/6] [CI/Build] Move OmniInteract cases to perf JSON Signed-off-by: natureofnature --- .buildkite/cuda/test-nightly.yml | 25 ++-- docs/cli/bench/serve.md | 11 +- tests/benchmarks/test_omniinteract.py | 4 +- tests/buildkite/test_upload_pipeline.py | 19 --- tests/dfx/perf/scripts/run_benchmark.py | 22 +++- .../tests/test_minicpmo_4_5_omniinteract.json | 90 +++++++++++++ tests/dfx/perf/tests/test_runner_metadata.py | 34 +++++ .../test_minicpmo_4_5_duplex_expansion.py | 120 ------------------ 8 files changed, 166 insertions(+), 159 deletions(-) create mode 100644 tests/dfx/perf/tests/test_minicpmo_4_5_omniinteract.json diff --git a/.buildkite/cuda/test-nightly.yml b/.buildkite/cuda/test-nightly.yml index fa9d1311c2b..b6f5f96c4ce 100644 --- a/.buildkite/cuda/test-nightly.yml +++ b/.buildkite/cuda/test-nightly.yml @@ -30,18 +30,6 @@ steps: - pytest -s -v tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py -m "full_model and cuda and H100 and omni and cards_1" --run-level "full_model" mirror_hardwares: h100_1 - - label: ":full_moon: Omni · MiniCPM-o 4.5 OmniInteract Nightly" - timeout_in_minutes: 300 - commands: - - export VLLM_OMNI_RUN_OMNIINTERACT_E2E=1 - - | - set +e - pytest -s -v tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py -k "omniinteract" -m "full_model and cuda and H100 and omni and cards_1" --run-level "full_model" --basetemp=tests/e2e/online_serving/omniinteract-nightly - EXIT=$$? - buildkite-agent artifact upload "tests/e2e/online_serving/omniinteract-nightly/**/*" || EXIT=1 - exit $$EXIT - mirror_hardwares: h100_1 - - label: ":full_moon: Omni · Doc Test with L4" timeout_in_minutes: 90 commands: @@ -120,6 +108,19 @@ steps: exit $$EXIT mirror_hardwares: h100_1 + - label: ":full_moon: Omni · MiniCPM-o 4.5 · OmniInteract Perf Test" + key: nightly-omni-performance-minicpmo-4-5-omniinteract + timeout_in_minutes: 300 + commands: + - export BENCHMARK_DIR=tests/dfx/perf/results + - | + set +e + pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minicpmo_4_5_omniinteract.json -m "full_model and cuda and H100 and omni and cards_1" + EXIT=$$? + buildkite-agent artifact upload "tests/dfx/perf/results/**/*" || EXIT=1 + exit $$EXIT + mirror_hardwares: h100_1 + - label: ":full_moon: Omni · MiniCPM-o 4.5 · Duplex Seed-TTS Perf Test" key: nightly-omni-performance-minicpmo-4-5-duplex-seed-tts timeout_in_minutes: 180 diff --git a/docs/cli/bench/serve.md b/docs/cli/bench/serve.md index de1f21d3eb4..c64f56bc465 100644 --- a/docs/cli/bench/serve.md +++ b/docs/cli/bench/serve.md @@ -307,11 +307,12 @@ omitted from the official manifest; `audio_clipped_bytes` records output beyond TTFT, TTFP, and RTF start at client receipt of `response.created`. TPOT/ITL use engine stage-0 timing; ITL is emitted only when every token interval is present. -The dedicated CUDA Nightly benchmark gate runs four deterministic cases from each OmniInteract subset (12 videos total), -with no warmups and a maximum concurrency of two. It requires every case to commit its input, complete any emitted response -lifecycles, and publish parseable WAV, transcript, event, and result artifacts. A valid LISTEN-only case may have no response -audio or transcript chunks. Official-manifest eligibility is reported separately because clipped or cancelled output is a -benchmark-quality signal, not a transport failure. This gate does not score answer accuracy. +The dedicated CUDA performance gate in the Nightly pipeline runs four deterministic cases from each OmniInteract subset +(12 videos total), with no warmups and a maximum concurrency of two. It requires every case to commit its input, complete +any emitted response lifecycles, and publish the expected WAV, transcript, event, and result artifacts without errors. A +valid LISTEN-only case may have no response audio or transcript chunks. Official-manifest eligibility is reported separately +because clipped or cancelled output is a benchmark-quality signal, not a transport failure. This gate does not score answer +accuracy. ### Multi-Modal Benchmark diff --git a/tests/benchmarks/test_omniinteract.py b/tests/benchmarks/test_omniinteract.py index 70fc8168055..6fc1365cca7 100644 --- a/tests/benchmarks/test_omniinteract.py +++ b/tests/benchmarks/test_omniinteract.py @@ -432,7 +432,7 @@ def test_atomic_write_preserves_destination_on_failure(tmp_path: Path, monkeypat class _RealtimeClient: instances: list[_RealtimeClient] = [] - def __init__(self, url: str, **kwargs: object): + def __init__(self, url: str, **kwargs): self.url = url self.events = RealtimeEventCollector() self.acks: list[tuple[str, int]] = [] @@ -445,7 +445,7 @@ async def __aenter__(self): async def __aexit__(self, *args): return None - async def configure(self, model: str, **kwargs: object) -> None: + async def configure(self, model: str, **kwargs) -> None: self.configure_kwargs = kwargs self.events.add({"type": "session.created", "session": {"capabilities": {"chunk_period_ms": 1000}}}) diff --git a/tests/buildkite/test_upload_pipeline.py b/tests/buildkite/test_upload_pipeline.py index c15194a8d77..c994f8dd29f 100644 --- a/tests/buildkite/test_upload_pipeline.py +++ b/tests/buildkite/test_upload_pipeline.py @@ -110,25 +110,6 @@ def test_yaml_gated_l45_only_does_not_unconditionally_build_image() -> None: assert "key: upload-weekly-pipeline" in rendered -def test_minicpmo_omniinteract_nightly_isolated_from_duplex() -> None: - path = Path(".buildkite/cuda/test-nightly.yml") - rendered = _render_test_pipeline(yaml.safe_load(path.read_text()), changed_files=None) - steps = [step for group in rendered["steps"] for step in group.get("steps", [])] - by_label = {step.get("label"): step for step in steps} - - duplex = by_label[":full_moon: Omni · MiniCPM-o 4.5 Duplex Test"] - assert duplex["timeout_in_minutes"] == 50 - assert len(duplex["commands"]) == 1 - assert "VLLM_OMNI_RUN_OMNIINTERACT_E2E" not in duplex["commands"][0] - - omniinteract = by_label[":full_moon: Omni · MiniCPM-o 4.5 OmniInteract Nightly"] - assert omniinteract["timeout_in_minutes"] == 300 - commands = "\n".join(omniinteract["commands"]) - assert "VLLM_OMNI_RUN_OMNIINTERACT_E2E=1" in commands - assert '-k "omniinteract"' in commands - assert "artifact upload" in commands - - def test_yaml_gated_l2_still_enables_image_via_ready_base() -> None: rendered = _render([".buildkite/cuda/test-ready.yml"]) assert 'build.pull_request.labels includes "ready"' in rendered diff --git a/tests/dfx/perf/scripts/run_benchmark.py b/tests/dfx/perf/scripts/run_benchmark.py index 64285dca36e..94df2edc55e 100644 --- a/tests/dfx/perf/scripts/run_benchmark.py +++ b/tests/dfx/perf/scripts/run_benchmark.py @@ -161,8 +161,26 @@ def benchmark_params(request): } +def _resolve_num_warmups(params: dict[str, Any], *, default: int) -> int: + value = params.get("num_warmups") + if value is None: + return default + if not isinstance(value, int) or isinstance(value, bool) or value < 0: + raise ValueError("num_warmups must be a non-negative integer") + return value + + def assert_result(result, params, num_prompt) -> None: assert result["completed"] == num_prompt, "Request failures exist" + if params.get("dataset_name") == "omniinteract": + summary = result.get("omniinteract") + assert isinstance(summary, dict), "OmniInteract summary is missing" + assert (summary.get("total"), summary.get("success"), summary.get("failed")) == ( + num_prompt, + num_prompt, + 0, + ), "OmniInteract requests did not all succeed" + assert summary.get("artifacts_complete") is True, "OmniInteract artifacts are incomplete" baseline = params.get("baseline") hardware = result.get("Hardware") hardware_baseline = baseline.get(hardware) if isinstance(baseline, dict) and isinstance(hardware, str) else None @@ -235,6 +253,7 @@ def to_list(value, default=None): "baseline", "num_prompts", "max_concurrency", + "num_warmups", "task", "enabled", "eval_phase", @@ -278,6 +297,7 @@ def to_list(value, default=None): random_input_len=params.get("random_input_len"), random_output_len=params.get("random_output_len"), resource_label=resource_label, + num_warmups=_resolve_num_warmups(params, default=2), ) assert_result(result, params, num_prompt) @@ -295,6 +315,6 @@ def to_list(value, default=None): random_input_len=params.get("random_input_len"), random_output_len=params.get("random_output_len"), resource_label=resource_label, - num_warmups=max(2, int(concurrency)), + num_warmups=_resolve_num_warmups(params, default=max(2, int(concurrency))), ) assert_result(result, params, num_prompt) diff --git a/tests/dfx/perf/tests/test_minicpmo_4_5_omniinteract.json b/tests/dfx/perf/tests/test_minicpmo_4_5_omniinteract.json new file mode 100644 index 00000000000..dccbbb65af5 --- /dev/null +++ b/tests/dfx/perf/tests/test_minicpmo_4_5_omniinteract.json @@ -0,0 +1,90 @@ +[ + { + "test_name": "test_minicpmo_4_5_omniinteract", + "mark": [ + { + "hardware_marks": { + "res": { + "cuda": "H100" + }, + "num_cards": 1 + } + }, + "full_model", + "omni" + ], + "server_params": { + "model": "openbmb/MiniCPM-o-4_5", + "extra_cli_args": [ + "--deploy-config", + "vllm_omni/deploy/minicpmo_4_5.yaml", + "--trust-remote-code" + ] + }, + "benchmark_params": [ + { + "dataset_name": "omniinteract", + "dataset_path": "lucky-lance/OmniInteract@e195f75fe2666fcc5fe74f537ae49ca143a79969", + "backend": "openai-realtime-duplex", + "endpoint": "/v1/realtime", + "num_prompts": [ + 4 + ], + "max_concurrency": [ + 2 + ], + "num_warmups": 0, + "no_oversample": true, + "disable_shuffle": true, + "trust_remote_code": true, + "omniinteract_subsets": "1q1a", + "omniinteract_ref_audio": "tests/assets/minicpmo_4_5/response_required_16k.wav", + "omniinteract_output_dir": "tests/dfx/perf/results/omniinteract/1q1a", + "omniinteract_max_video_duration_s": 600, + "percentile-metrics": "ttft,tpot,itl,e2el,audio_ttfp,audio_rtf,audio_duration" + }, + { + "dataset_name": "omniinteract", + "dataset_path": "lucky-lance/OmniInteract@e195f75fe2666fcc5fe74f537ae49ca143a79969", + "backend": "openai-realtime-duplex", + "endpoint": "/v1/realtime", + "num_prompts": [ + 4 + ], + "max_concurrency": [ + 2 + ], + "num_warmups": 0, + "no_oversample": true, + "disable_shuffle": true, + "trust_remote_code": true, + "omniinteract_subsets": "1q1a_math", + "omniinteract_ref_audio": "tests/assets/minicpmo_4_5/response_required_16k.wav", + "omniinteract_output_dir": "tests/dfx/perf/results/omniinteract/1q1a_math", + "omniinteract_max_video_duration_s": 600, + "percentile-metrics": "ttft,tpot,itl,e2el,audio_ttfp,audio_rtf,audio_duration" + }, + { + "dataset_name": "omniinteract", + "dataset_path": "lucky-lance/OmniInteract@e195f75fe2666fcc5fe74f537ae49ca143a79969", + "backend": "openai-realtime-duplex", + "endpoint": "/v1/realtime", + "num_prompts": [ + 4 + ], + "max_concurrency": [ + 2 + ], + "num_warmups": 0, + "no_oversample": true, + "disable_shuffle": true, + "trust_remote_code": true, + "omniinteract_subsets": "1qna", + "omniinteract_ref_audio": "tests/assets/minicpmo_4_5/response_required_16k.wav", + "omniinteract_output_dir": "tests/dfx/perf/results/omniinteract/1qna", + "omniinteract_max_video_duration_s": 600, + "percentile-metrics": "ttft,tpot,itl,e2el,audio_ttfp,audio_rtf,audio_duration" + } + ] + } +] diff --git a/tests/dfx/perf/tests/test_runner_metadata.py b/tests/dfx/perf/tests/test_runner_metadata.py index 2e98dcaf76b..17a040135fd 100644 --- a/tests/dfx/perf/tests/test_runner_metadata.py +++ b/tests/dfx/perf/tests/test_runner_metadata.py @@ -517,6 +517,40 @@ def test_omni_duplex_expected_audio_turns_rejects_incomplete_session(): ) +def test_num_warmups_preserves_explicit_zero(): + from tests.dfx.perf.scripts.run_benchmark import _resolve_num_warmups + + assert _resolve_num_warmups({}, default=4) == 4 + assert _resolve_num_warmups({"num_warmups": 0}, default=4) == 0 + + +def test_omniinteract_result_accepts_complete_artifacts(): + from tests.dfx.perf.scripts.run_benchmark import assert_result + + assert_result( + { + "completed": 4, + "omniinteract": {"total": 4, "success": 4, "failed": 0, "artifacts_complete": True}, + }, + {"dataset_name": "omniinteract"}, + 4, + ) + + +def test_omniinteract_result_rejects_incomplete_artifacts(): + from tests.dfx.perf.scripts.run_benchmark import assert_result + + with pytest.raises(AssertionError, match="artifacts are incomplete"): + assert_result( + { + "completed": 4, + "omniinteract": {"total": 4, "success": 4, "failed": 0, "artifacts_complete": False}, + }, + {"dataset_name": "omniinteract"}, + 4, + ) + + def test_omni_tpot_baseline_accepts_measured_finite_sample(): from tests.dfx.perf.scripts.run_benchmark import assert_result diff --git a/tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py b/tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py index edcbae099f5..5b35c4b53fe 100644 --- a/tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py +++ b/tests/e2e/online_serving/test_minicpmo_4_5_duplex_expansion.py @@ -7,10 +7,6 @@ import asyncio import json -import os -import shutil -import subprocess -import wave from pathlib import Path from types import SimpleNamespace @@ -34,12 +30,9 @@ run_soft_interrupt, ) from tests.helpers.mark import hardware_test -from vllm_omni.benchmarks.data_modules.omniinteract_dataset import DEFAULT_OMNIINTERACT_REPO pytestmark = [pytest.mark.full_model, pytest.mark.omni] -OMNIINTERACT_NIGHTLY_DATASET = f"{DEFAULT_OMNIINTERACT_REPO}@e195f75fe2666fcc5fe74f537ae49ca143a79969" - @hardware_test(res={"cuda": "H100", "npu": "A3"}, num_cards=1) @pytest.mark.parametrize("omni_server", SERVER_PARAMS, indirect=True) @@ -119,116 +112,3 @@ def test_duplex_server_vad_hard_interrupt(omni_server) -> None: ) ) assert result["ok"] is True, json.dumps(result, ensure_ascii=False, indent=2) - - -@hardware_test(res={"cuda": "H100"}, num_cards=1) -@pytest.mark.skipif( - os.environ.get("VLLM_OMNI_RUN_OMNIINTERACT_E2E") != "1", - reason="enable the real-time OmniInteract dataset E2E", -) -@pytest.mark.parametrize("subset", ("1q1a", "1q1a_math", "1qna")) -@pytest.mark.parametrize("omni_server", SERVER_PARAMS, indirect=True) -def test_minicpmo_4_5_omniinteract_nightly( - omni_server, - tmp_path: Path, - subset: str, -) -> None: - dataset_path = os.environ.get("OMNIINTERACT_ROOT", "").strip() or OMNIINTERACT_NIGHTLY_DATASET - run_root = tmp_path / subset - artifact_root = run_root / "artifacts" - result_filename = "benchmark_result.json" - vllm = shutil.which("vllm") - assert vllm is not None, "Could not find `vllm` on PATH" - subprocess.run( - [ - vllm, - "bench", - "serve", - "--omni", - "--trust-remote-code", - "--host", - omni_server.host, - "--port", - str(omni_server.port), - "--backend", - "openai-realtime-duplex", - "--endpoint", - "/v1/realtime", - "--model", - omni_server.model, - "--dataset-name", - "omniinteract", - "--dataset-path", - dataset_path, - "--omniinteract-subsets", - subset, - "--omniinteract-ref-audio", - str(resolve_ref_audio()), - "--omniinteract-output-dir", - str(artifact_root), - "--omniinteract-max-video-duration-s", - "600", - "--num-warmups", - "0", - "--num-prompts", - "4", - "--max-concurrency", - "2", - "--request-rate", - "inf", - "--disable-shuffle", - "--save-result", - "--result-dir", - str(run_root), - "--result-filename", - result_filename, - ], - check=True, - ) - - benchmark_result = json.loads((run_root / result_filename).read_text()) - artifact_summary = benchmark_result["omniinteract"] - assert artifact_summary["artifacts_complete"] is True - assert (artifact_summary["total"], artifact_summary["success"], artifact_summary["failed"]) == (4, 4, 0) - - batch_summary = json.loads((artifact_root / "batch_summary.json").read_text()) - assert (batch_summary["total"], batch_summary["success"], batch_summary["failed"]) == (4, 4, 0) - manifest_rows = [ - json.loads(row) for row in (artifact_root / "official_eval_manifest.jsonl").read_text().splitlines() - ] - assert len(manifest_rows) == batch_summary["eligible_for_official_eval"] - assert {row["video"] for row in manifest_rows} == { - result["video"] for result in batch_summary["results"] if result["eligible_for_official_eval"] - } - assert {result["subset"] for result in batch_summary["results"]} == {subset} - assert len({result["video"] for result in batch_summary["results"]}) == 4 - - for result in batch_summary["results"]: - assert result["success"] is True - assert result["input_audio_chunks"] > 0 - assert result["input_video_frames"] > 0 - sample_root = Path(result["output_dir"]) - assert all( - (sample_root / name).is_file() - for name in (".done", "output.wav", "wav_transcript.json", "events.json", "result.json") - ) - with wave.open(str(sample_root / "output.wav"), "rb") as output_wav: - assert ( - output_wav.getframerate(), - output_wav.getnchannels(), - output_wav.getsampwidth(), - output_wav.getcomptype(), - ) == (24_000, 1, 2, "NONE") - transcript = json.loads((sample_root / "wav_transcript.json").read_text()) - assert isinstance(transcript["chunks"], list) - assert transcript["timestamp_semantics"] - published_result = json.loads((sample_root / "result.json").read_text()) - assert published_result["video"] == result["video"] - assert published_result["success"] is True - events = json.loads((sample_root / "events.json").read_text()) - created = {event["response"]["id"] for event in events if event.get("type") == "response.created"} - done = {event["response"]["id"] for event in events if event.get("type") == "response.done"} - assert created == done - assert len(created) == result["responses"] - if not result["eligible_for_official_eval"]: - assert result["official_eval_ineligible_reasons"] From 8b2b6095e1b4af2aa95c6d10243274c4aed6e6a1 Mon Sep 17 00:00:00 2001 From: natureofnature Date: Wed, 2 Sep 2026 04:56:29 -0400 Subject: [PATCH 5/6] [Benchmark] Keep OmniInteract perf test local Signed-off-by: natureofnature --- .buildkite/cuda/test-nightly.yml | 13 ------ docs/cli/bench/serve.md | 25 ++++++++--- tests/benchmarks/test_omniinteract.py | 8 +--- .../tests/test_minicpmo_4_5_omniinteract.json | 5 +-- tests/tools/test_run_nightly_jobs.py | 43 +++++++++++++++++++ tools/nightly/run_nightly_jobs.sh | 2 +- 6 files changed, 66 insertions(+), 30 deletions(-) create mode 100644 tests/tools/test_run_nightly_jobs.py diff --git a/.buildkite/cuda/test-nightly.yml b/.buildkite/cuda/test-nightly.yml index b6f5f96c4ce..9f44ba2b0ce 100644 --- a/.buildkite/cuda/test-nightly.yml +++ b/.buildkite/cuda/test-nightly.yml @@ -108,19 +108,6 @@ steps: exit $$EXIT mirror_hardwares: h100_1 - - label: ":full_moon: Omni · MiniCPM-o 4.5 · OmniInteract Perf Test" - key: nightly-omni-performance-minicpmo-4-5-omniinteract - timeout_in_minutes: 300 - commands: - - export BENCHMARK_DIR=tests/dfx/perf/results - - | - set +e - pytest -s -v tests/dfx/perf/scripts/run_benchmark.py --test-config-file tests/dfx/perf/tests/test_minicpmo_4_5_omniinteract.json -m "full_model and cuda and H100 and omni and cards_1" - EXIT=$$? - buildkite-agent artifact upload "tests/dfx/perf/results/**/*" || EXIT=1 - exit $$EXIT - mirror_hardwares: h100_1 - - label: ":full_moon: Omni · MiniCPM-o 4.5 · Duplex Seed-TTS Perf Test" key: nightly-omni-performance-minicpmo-4-5-duplex-seed-tts timeout_in_minutes: 180 diff --git a/docs/cli/bench/serve.md b/docs/cli/bench/serve.md index c64f56bc465..efb23df00c9 100644 --- a/docs/cli/bench/serve.md +++ b/docs/cli/bench/serve.md @@ -307,12 +307,25 @@ omitted from the official manifest; `audio_clipped_bytes` records output beyond TTFT, TTFP, and RTF start at client receipt of `response.created`. TPOT/ITL use engine stage-0 timing; ITL is emitted only when every token interval is present. -The dedicated CUDA performance gate in the Nightly pipeline runs four deterministic cases from each OmniInteract subset -(12 videos total), with no warmups and a maximum concurrency of two. It requires every case to commit its input, complete -any emitted response lifecycles, and publish the expected WAV, transcript, event, and result artifacts without errors. A -valid LISTEN-only case may have no response audio or transcript chunks. Official-manifest eligibility is reported separately -because clipped or cancelled output is a benchmark-quality signal, not a transport failure. This gate does not score answer -accuracy. +The checked-in local performance configuration measures four deterministic cases from each OmniInteract subset (12 videos +total), with no benchmark warmups and a maximum concurrency of two. Each subset also sends one readiness request before its +measured cases. Run it from the repository root with: + +```bash +export HF_HOME=/path/to/persistent/huggingface-cache +export BENCHMARK_DIR=tests/dfx/perf/results +bash tools/nightly/run_nightly_jobs.sh \ + --test-type local \ + --model-type omni \ + --label-substr minicpmo_4_5_omniinteract +``` + +The first run downloads the pinned OmniInteract archive into `HF_HOME`; later runs reuse that cache. + +It requires every case to commit its input, complete any emitted response lifecycles, and publish the expected WAV, +transcript, event, and result artifacts without errors. A valid LISTEN-only case may have no response audio or transcript +chunks. Official-manifest eligibility is reported separately because clipped or cancelled output is a benchmark-quality +signal, not a transport failure. This local performance test does not score answer accuracy. ### Multi-Modal Benchmark diff --git a/tests/benchmarks/test_omniinteract.py b/tests/benchmarks/test_omniinteract.py index 6fc1365cca7..c51724e0a8a 100644 --- a/tests/benchmarks/test_omniinteract.py +++ b/tests/benchmarks/test_omniinteract.py @@ -286,8 +286,7 @@ def test_response_ledger_rejects_identity_errors(events, match: str): class _CompletionClient: def __init__(self, collector: RealtimeEventCollector): - self.events = collector - self.acks: list[tuple[str, int]] = [] + self.events, self.acks = collector, [] def raise_if_reader_stopped(self) -> None: return None @@ -433,10 +432,7 @@ class _RealtimeClient: instances: list[_RealtimeClient] = [] def __init__(self, url: str, **kwargs): - self.url = url - self.events = RealtimeEventCollector() - self.acks: list[tuple[str, int]] = [] - self.configure_kwargs: dict[str, object] = {} + self.url, self.events, self.acks, self.configure_kwargs = url, RealtimeEventCollector(), [], {} self.instances.append(self) async def __aenter__(self): diff --git a/tests/dfx/perf/tests/test_minicpmo_4_5_omniinteract.json b/tests/dfx/perf/tests/test_minicpmo_4_5_omniinteract.json index dccbbb65af5..f101fd5155d 100644 --- a/tests/dfx/perf/tests/test_minicpmo_4_5_omniinteract.json +++ b/tests/dfx/perf/tests/test_minicpmo_4_5_omniinteract.json @@ -10,7 +10,7 @@ "num_cards": 1 } }, - "full_model", + "local_model", "omni" ], "server_params": { @@ -34,7 +34,6 @@ 2 ], "num_warmups": 0, - "no_oversample": true, "disable_shuffle": true, "trust_remote_code": true, "omniinteract_subsets": "1q1a", @@ -55,7 +54,6 @@ 2 ], "num_warmups": 0, - "no_oversample": true, "disable_shuffle": true, "trust_remote_code": true, "omniinteract_subsets": "1q1a_math", @@ -76,7 +74,6 @@ 2 ], "num_warmups": 0, - "no_oversample": true, "disable_shuffle": true, "trust_remote_code": true, "omniinteract_subsets": "1qna", diff --git a/tests/tools/test_run_nightly_jobs.py b/tests/tools/test_run_nightly_jobs.py new file mode 100644 index 00000000000..d7ccc823406 --- /dev/null +++ b/tests/tools/test_run_nightly_jobs.py @@ -0,0 +1,43 @@ +# SPDX-License-Identifier: Apache-2.0 +# SPDX-FileCopyrightText: Copyright contributors to the vLLM project + +import os +import subprocess +from pathlib import Path + +import pytest + +pytestmark = [pytest.mark.core_model, pytest.mark.cpu] + + +def test_local_minicpmo_perf_uses_omni_runner(tmp_path: Path): + repo_root = Path(__file__).resolve().parents[2] + script = repo_root / "tools" / "nightly" / "run_nightly_jobs.sh" + env = os.environ.copy() + env.update({"REPO_ROOT": str(repo_root), "LOG_DIR": str(tmp_path / "logs")}) + + result = subprocess.run( + [ + "bash", + str(script), + "--test-type", + "local", + "--model-type", + "omni", + "--label-substr", + "minicpmo_4_5_omniinteract", + "--dry-run", + ], + env=env, + capture_output=True, + text=True, + timeout=30, + check=False, + ) + output = result.stdout + result.stderr + + assert result.returncode == 0, output + assert "omni and local_model" in output + assert "tests/dfx/perf/scripts/run_benchmark.py" in output + assert "test_minicpmo_4_5_omniinteract.json" in output + assert "run_diffusion_benchmark.py" not in output diff --git a/tools/nightly/run_nightly_jobs.sh b/tools/nightly/run_nightly_jobs.sh index b5b6b92ce98..73f6b5f71ae 100644 --- a/tools/nightly/run_nightly_jobs.sh +++ b/tools/nightly/run_nightly_jobs.sh @@ -678,7 +678,7 @@ TTS_PERF_JSON_HINTS = ("test_tts", "voxcpm", "higgs_audio") def perf_json_model_family(json_basename: str) -> str: """Classify a perf JSON config as omni, tts, or diffusion (mirrors nightly YAML runners).""" name = json_basename.lower() - if name.startswith("test_qwen3_omni"): + if name.startswith(("test_qwen3_omni", "test_minicpmo_")): return "omni" if any(hint in name for hint in TTS_PERF_JSON_HINTS): return "tts" From e50be2da737c9115f41f7294050145d1120beab1 Mon Sep 17 00:00:00 2001 From: natureofnature Date: Thu, 3 Sep 2026 08:58:47 -0400 Subject: [PATCH 6/6] [Test] Remove redundant local launcher coverage Signed-off-by: natureofnature --- tests/tools/test_run_nightly_jobs.py | 43 ---------------------------- 1 file changed, 43 deletions(-) delete mode 100644 tests/tools/test_run_nightly_jobs.py diff --git a/tests/tools/test_run_nightly_jobs.py b/tests/tools/test_run_nightly_jobs.py deleted file mode 100644 index d7ccc823406..00000000000 --- a/tests/tools/test_run_nightly_jobs.py +++ /dev/null @@ -1,43 +0,0 @@ -# SPDX-License-Identifier: Apache-2.0 -# SPDX-FileCopyrightText: Copyright contributors to the vLLM project - -import os -import subprocess -from pathlib import Path - -import pytest - -pytestmark = [pytest.mark.core_model, pytest.mark.cpu] - - -def test_local_minicpmo_perf_uses_omni_runner(tmp_path: Path): - repo_root = Path(__file__).resolve().parents[2] - script = repo_root / "tools" / "nightly" / "run_nightly_jobs.sh" - env = os.environ.copy() - env.update({"REPO_ROOT": str(repo_root), "LOG_DIR": str(tmp_path / "logs")}) - - result = subprocess.run( - [ - "bash", - str(script), - "--test-type", - "local", - "--model-type", - "omni", - "--label-substr", - "minicpmo_4_5_omniinteract", - "--dry-run", - ], - env=env, - capture_output=True, - text=True, - timeout=30, - check=False, - ) - output = result.stdout + result.stderr - - assert result.returncode == 0, output - assert "omni and local_model" in output - assert "tests/dfx/perf/scripts/run_benchmark.py" in output - assert "test_minicpmo_4_5_omniinteract.json" in output - assert "run_diffusion_benchmark.py" not in output