Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 0 additions & 1 deletion benchmarks/single_node/agentic/minimaxm3_fp4_b200_mtp.sh
Original file line number Diff line number Diff line change
Expand Up @@ -126,7 +126,6 @@ install_agentic_deps

OFFLOAD_ARGS=()
if require_agentic_kv_offload_backend vllm-simple; then
python3 "$(dirname "$0")/../../../runners/patch_vllm_simple_kv_offload.py"
CPU_OFFLOAD_BYTES=$((TOTAL_CPU_DRAM_GB * 1024 * 1024 * 1024))
export VLLM_USE_SIMPLE_KV_OFFLOAD=1
OFFLOAD_CONFIG=$(printf \
Expand Down
2 changes: 1 addition & 1 deletion configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -7983,7 +7983,7 @@ minimaxm3-fp4-b300-trtllm-agentic-mtp:
# the same 3 TB AgentX ceiling before the proportional-GPU rule is applied.
# GPU-resident points receive a zero budget.
minimaxm3-fp4-b200-vllm-agentic-mtp:
image: vllm/vllm-openai:nightly-1dc464d42681d22f38caf1fdc1eb632dc4421c45
image: vllm/vllm-openai:nightly-cd10ed6f9f6b37a8ace9cf380007e66fe12ec0c3
model: nvidia/MiniMax-M3-NVFP4
model-prefix: minimaxm3
runner: cluster:b200-nscale
Expand Down
8 changes: 8 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -7807,3 +7807,11 @@
- "Use FLASH_ATTN for the target model, select the Humming MXFP8 dense-linear backend, and enable Marlin atomic reduction for the MXFP8 MoE path."
- "Use 0.95 GPU memory utilization for resident serving while retaining 0.90 for Mooncake DRAM offload, and remove the Mooncake concurrency-14 point."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3108

- config-keys:
- minimaxm3-fp4-b200-vllm-agentic-mtp
scenario-type:
- agentic-coding
description:
- "Update B200 MiniMax-M3 AgentX to the latest vLLM image with upstream mixed-page KV offload support and remove the runtime offload patch."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3131
133 changes: 0 additions & 133 deletions runners/patch_vllm_simple_kv_offload.py

This file was deleted.

47 changes: 0 additions & 47 deletions runners/test_slurm_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,6 @@
PATCH_SRT_EVAL = REPO_ROOT / "runners" / "patch_srt_eval_dispatch.py"
PATCH_SRT_DP_RANKS = REPO_ROOT / "runners" / "patch_srt_vllm_dp_ranks.py"
PATCH_TRTLLM_CHAT_STORE = REPO_ROOT / "runners" / "patch_trtllm_chat_store.py"
PATCH_VLLM_SIMPLE_KV = REPO_ROOT / "runners" / "patch_vllm_simple_kv_offload.py"
INJECT_ACCEPTANCE = REPO_ROOT / "runners" / "inject_synthetic_acceptance.py"


Expand Down Expand Up @@ -393,52 +392,6 @@ def test_patch_trtllm_chat_store_rejects_unknown_source(tmp_path: Path) -> None:
assert result.returncode == 1
assert protocol.read_text() == "unsupported protocol\n"

def test_patch_vllm_simple_kv_offload_is_idempotent_and_preserves_surrounding_code(
tmp_path: Path,
) -> None:
symbols = runpy.run_path(str(PATCH_VLLM_SIMPLE_KV))
worker = tmp_path / "worker.py"
original = f"prefix\n{symbols['OLD_SETUP']}{symbols['OLD_LOOP']}suffix\n"
worker.write_text(original)

first = subprocess.run(
["python3", str(PATCH_VLLM_SIMPLE_KV), str(worker)],
check=False,
capture_output=True,
text=True,
)
patched = worker.read_text()
second = subprocess.run(
["python3", str(PATCH_VLLM_SIMPLE_KV), str(worker)],
check=False,
capture_output=True,
text=True,
)

assert first.returncode == 0, first.stderr
assert second.returncode == 0, second.stderr
assert patched != original
assert patched.startswith("prefix\n") and patched.endswith("suffix\n")
assert worker.read_text() == patched


def test_patch_vllm_simple_kv_offload_rejects_unknown_source(
tmp_path: Path,
) -> None:
worker = tmp_path / "worker.py"
worker.write_text("unsupported worker\n")

result = subprocess.run(
["python3", str(PATCH_VLLM_SIMPLE_KV), str(worker)],
check=False,
capture_output=True,
text=True,
)

assert result.returncode == 1
assert worker.read_text() == "unsupported worker\n"


def test_patch_srt_eval_dispatch_preflights_before_writing(tmp_path: Path) -> None:
do_sweep = tmp_path / "src/srtctl/cli/do_sweep.py"
eval_script = tmp_path / "src/srtctl/benchmarks/scripts/lm-eval/bench.sh"
Expand Down
Loading