Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 20 additions & 0 deletions docs/cli/bench/serve.md
Original file line number Diff line number Diff line change
Expand Up @@ -307,6 +307,26 @@ omitted from the official manifest; `audio_clipped_bytes` records output beyond
TTFT, TTFP, and RTF start at client receipt of `response.created`. TPOT/ITL use engine stage-0 timing; ITL is emitted only when
every token interval is present.

The checked-in local performance configuration measures four deterministic cases from each OmniInteract subset (12 videos
total), with no benchmark warmups and a maximum concurrency of two. Each subset also sends one readiness request before its
measured cases. Run it from the repository root with:

```bash
export HF_HOME=/path/to/persistent/huggingface-cache
export BENCHMARK_DIR=tests/dfx/perf/results
bash tools/nightly/run_nightly_jobs.sh \
--test-type local \
--model-type omni \
--label-substr minicpmo_4_5_omniinteract
```

The first run downloads the pinned OmniInteract archive into `HF_HOME`; later runs reuse that cache.

It requires every case to commit its input, complete any emitted response lifecycles, and publish the expected WAV,
transcript, event, and result artifacts without errors. A valid LISTEN-only case may have no response audio or transcript
chunks. Official-manifest eligibility is reported separately because clipped or cancelled output is a benchmark-quality
signal, not a transport failure. This local performance test does not score answer accuracy.

### Multi-Modal Benchmark

<details class="admonition abstract" markdown="1">
Expand Down
7 changes: 4 additions & 3 deletions tests/benchmarks/test_omniinteract.py
Original file line number Diff line number Diff line change
Expand Up @@ -203,18 +203,19 @@ def test_archive_is_safe_and_atomically_shared(tmp_path: Path, monkeypatch: pyte
assert not list(target.glob(".tmp-*"))


def test_hub_archive_uses_vllm_filesystem(tmp_path: Path, monkeypatch: pytest.MonkeyPatch):
@pytest.mark.parametrize("dataset_repo", ["org/repo", "org/repo@revision"])
def test_hub_archive_uses_vllm_filesystem(tmp_path: Path, monkeypatch: pytest.MonkeyPatch, dataset_repo: str):
source = tmp_path / "source.tar"
_archive(source, "data/1q1a/video_json_map.json")

class FS:
def get_file(self, remote: str, local: str) -> None:
assert remote == "datasets/org/repo/data.tar.gz"
assert remote == f"datasets/{dataset_repo}/data.tar.gz"
Path(local).write_bytes(source.read_bytes())

monkeypatch.setenv("HF_HOME", str(tmp_path / "hf"))
monkeypatch.setattr(data, "hf_fs", lambda: FS())
assert (data.resolve_omniinteract_root(None, "org/repo") / "1q1a").is_dir()
assert (data.resolve_omniinteract_root(None, dataset_repo) / "1q1a").is_dir()


def test_media_commands_are_bounded(monkeypatch: pytest.MonkeyPatch, tmp_path: Path):
Expand Down
22 changes: 21 additions & 1 deletion tests/dfx/perf/scripts/run_benchmark.py
Original file line number Diff line number Diff line change
Expand Up @@ -161,8 +161,26 @@ def benchmark_params(request):
}


def _resolve_num_warmups(params: dict[str, Any], *, default: int) -> int:

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This function is kind of non-negative integer validating helper, which could be centralized to metrics/utils.py later if several modules need it.

value = params.get("num_warmups")
if value is None:
return default
if not isinstance(value, int) or isinstance(value, bool) or value < 0:
raise ValueError("num_warmups must be a non-negative integer")
return value


def assert_result(result, params, num_prompt) -> None:
assert result["completed"] == num_prompt, "Request failures exist"
if params.get("dataset_name") == "omniinteract":
summary = result.get("omniinteract")
assert isinstance(summary, dict), "OmniInteract summary is missing"
assert (summary.get("total"), summary.get("success"), summary.get("failed")) == (
num_prompt,
num_prompt,
0,
), "OmniInteract requests did not all succeed"
assert summary.get("artifacts_complete") is True, "OmniInteract artifacts are incomplete"
baseline = params.get("baseline")
hardware = result.get("Hardware")
hardware_baseline = baseline.get(hardware) if isinstance(baseline, dict) and isinstance(hardware, str) else None
Expand Down Expand Up @@ -235,6 +253,7 @@ def to_list(value, default=None):
"baseline",
"num_prompts",
"max_concurrency",
"num_warmups",
"task",
"enabled",
"eval_phase",
Expand Down Expand Up @@ -278,6 +297,7 @@ def to_list(value, default=None):
random_input_len=params.get("random_input_len"),
random_output_len=params.get("random_output_len"),
resource_label=resource_label,
num_warmups=_resolve_num_warmups(params, default=2),
)
assert_result(result, params, num_prompt)

Expand All @@ -295,6 +315,6 @@ def to_list(value, default=None):
random_input_len=params.get("random_input_len"),
random_output_len=params.get("random_output_len"),
resource_label=resource_label,
num_warmups=max(2, int(concurrency)),
num_warmups=_resolve_num_warmups(params, default=max(2, int(concurrency))),
)
assert_result(result, params, num_prompt)
87 changes: 87 additions & 0 deletions tests/dfx/perf/tests/test_minicpmo_4_5_omniinteract.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,87 @@
[
{
"test_name": "test_minicpmo_4_5_omniinteract",
"mark": [
{
"hardware_marks": {
"res": {
"cuda": "H100"
},
"num_cards": 1
}
},
"local_model",
"omni"
],
"server_params": {
"model": "openbmb/MiniCPM-o-4_5",
"extra_cli_args": [
"--deploy-config",
"vllm_omni/deploy/minicpmo_4_5.yaml",
"--trust-remote-code"
]
},
"benchmark_params": [
{
"dataset_name": "omniinteract",
"dataset_path": "lucky-lance/OmniInteract@e195f75fe2666fcc5fe74f537ae49ca143a79969",
"backend": "openai-realtime-duplex",
"endpoint": "/v1/realtime",
"num_prompts": [
4
],
"max_concurrency": [
2
],
"num_warmups": 0,
"disable_shuffle": true,
"trust_remote_code": true,
"omniinteract_subsets": "1q1a",
"omniinteract_ref_audio": "tests/assets/minicpmo_4_5/response_required_16k.wav",
"omniinteract_output_dir": "tests/dfx/perf/results/omniinteract/1q1a",
"omniinteract_max_video_duration_s": 600,
"percentile-metrics": "ttft,tpot,itl,e2el,audio_ttfp,audio_rtf,audio_duration"
},
{
"dataset_name": "omniinteract",
"dataset_path": "lucky-lance/OmniInteract@e195f75fe2666fcc5fe74f537ae49ca143a79969",
"backend": "openai-realtime-duplex",
"endpoint": "/v1/realtime",
"num_prompts": [
4
],
"max_concurrency": [
2
],
"num_warmups": 0,
"disable_shuffle": true,
"trust_remote_code": true,
"omniinteract_subsets": "1q1a_math",
"omniinteract_ref_audio": "tests/assets/minicpmo_4_5/response_required_16k.wav",
"omniinteract_output_dir": "tests/dfx/perf/results/omniinteract/1q1a_math",
"omniinteract_max_video_duration_s": 600,
"percentile-metrics": "ttft,tpot,itl,e2el,audio_ttfp,audio_rtf,audio_duration"
},
{
"dataset_name": "omniinteract",
"dataset_path": "lucky-lance/OmniInteract@e195f75fe2666fcc5fe74f537ae49ca143a79969",
"backend": "openai-realtime-duplex",
"endpoint": "/v1/realtime",
"num_prompts": [
4
],
"max_concurrency": [
2
],
"num_warmups": 0,
"disable_shuffle": true,
"trust_remote_code": true,
"omniinteract_subsets": "1qna",
"omniinteract_ref_audio": "tests/assets/minicpmo_4_5/response_required_16k.wav",
"omniinteract_output_dir": "tests/dfx/perf/results/omniinteract/1qna",
"omniinteract_max_video_duration_s": 600,
"percentile-metrics": "ttft,tpot,itl,e2el,audio_ttfp,audio_rtf,audio_duration"
}
]
}
]
34 changes: 34 additions & 0 deletions tests/dfx/perf/tests/test_runner_metadata.py
Original file line number Diff line number Diff line change
Expand Up @@ -518,6 +518,40 @@ def test_omni_duplex_expected_audio_turns_rejects_incomplete_session():
)


def test_num_warmups_preserves_explicit_zero():
from tests.dfx.perf.scripts.run_benchmark import _resolve_num_warmups

assert _resolve_num_warmups({}, default=4) == 4
assert _resolve_num_warmups({"num_warmups": 0}, default=4) == 0


def test_omniinteract_result_accepts_complete_artifacts():
from tests.dfx.perf.scripts.run_benchmark import assert_result

assert_result(
{
"completed": 4,
"omniinteract": {"total": 4, "success": 4, "failed": 0, "artifacts_complete": True},
},
{"dataset_name": "omniinteract"},
4,
)


def test_omniinteract_result_rejects_incomplete_artifacts():
from tests.dfx.perf.scripts.run_benchmark import assert_result

with pytest.raises(AssertionError, match="artifacts are incomplete"):
assert_result(
{
"completed": 4,
"omniinteract": {"total": 4, "success": 4, "failed": 0, "artifacts_complete": False},
},
{"dataset_name": "omniinteract"},
4,
)


def test_omni_tpot_baseline_accepts_measured_finite_sample():
from tests.dfx.perf.scripts.run_benchmark import assert_result

Expand Down
2 changes: 1 addition & 1 deletion tools/nightly/run_nightly_jobs.sh
Original file line number Diff line number Diff line change
Expand Up @@ -678,7 +678,7 @@ TTS_PERF_JSON_HINTS = ("test_tts", "voxcpm", "higgs_audio")
def perf_json_model_family(json_basename: str) -> str:
"""Classify a perf JSON config as omni, tts, or diffusion (mirrors nightly YAML runners)."""
name = json_basename.lower()
if name.startswith("test_qwen3_omni"):
if name.startswith(("test_qwen3_omni", "test_minicpmo_")):
return "omni"
if any(hint in name for hint in TTS_PERF_JSON_HINTS):
return "tts"
Expand Down
Loading