From d71235dfcc10a6dd35e1627a824f8193b4e5affa Mon Sep 17 00:00:00 2001 From: Tyrone <71038642+TyroneNel@users.noreply.github.com> Date: Mon, 21 Sep 2026 21:10:20 +0000 Subject: [PATCH] bench: pass the served model name, keep the checkpoint as --tokenizer vllm bench serve posts --model verbatim as the request's model in its /tokenize alignment probe, so passing the checkpoint path 404s the model check there and every run logs "WARNING: /tokenize unavailable, skipping alignment" (314 of 602 campaign logs carry it). --model is now the served name (qwen3.8-27b) on every bench client invocation, and --tokenizer loads the tokenizer from the checkpoint dir, which is what the path was for. Zero risk, no server restart: requests already carried the served name via --served-model-name, only the probe was fed the path. No existing row is invalidated, so nothing needs re-running: the random dataset re-encodes prompts with the same local tokenizer either way, the alignment step the warning skips would have early-returned on token-count agreement, 0 of 602 logs show the client's own tokenizer-mismatch warning, and the custom cohorts never reach the probe. The campaign tables stand. --- bench/prefill_ab.sh | 5 ++++- bench/real_rep.sh | 4 +++- bench/run_benchmarks.sh | 7 ++++++- bench/warmup.sh | 5 ++++- 4 files changed, 17 insertions(+), 4 deletions(-) diff --git a/bench/prefill_ab.sh b/bench/prefill_ab.sh index cc62f2fd..10469d1a 100755 --- a/bench/prefill_ab.sh +++ b/bench/prefill_ab.sh @@ -20,7 +20,10 @@ OUT="$HERE/results-prefill-ab/$ARM"; mkdir -p "$OUT" export PATH="$REPO/venv/bin:$PATH" export OPENAI_API_KEY=${VLLM_API_KEY:-$(cat "$REPO/api_key.txt" 2>/dev/null)} MODEL=${MODEL:-$REPO/models/Qwen3.8-27B-W4A16-AutoRound-fast} -B="venv/bin/vllm bench serve --host 127.0.0.1 --port $PORT --model $MODEL --served-model-name qwen3.8-27b" +# --model is the served name (the bench client's /tokenize alignment probe +# posts it as the request's model; a checkpoint path 404s there); the +# checkpoint dir rides --tokenizer, which is what actually reads it. +B="venv/bin/vllm bench serve --host 127.0.0.1 --port $PORT --model qwen3.8-27b --tokenizer $MODEL --served-model-name qwen3.8-27b" # ---- boot ------------------------------------------------------------------- if curl -sf -o /dev/null http://127.0.0.1:$PORT/health; then diff --git a/bench/real_rep.sh b/bench/real_rep.sh index 0b38e510..e7b77f7e 100644 --- a/bench/real_rep.sh +++ b/bench/real_rep.sh @@ -8,7 +8,9 @@ HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"; REPO="$(dirname "$HERE")"; export PATH="$REPO/venv/bin:$PATH" export OPENAI_API_KEY=${VLLM_API_KEY:-$(cat "$REPO/api_key.txt" 2>/dev/null)} M=${MODEL:-$REPO/models/Qwen3.8-27B-W4A16-AutoRound}; TAG=$1; N=${2:-3}; T=${3:-} -B="venv/bin/vllm bench serve --host 127.0.0.1 --port 18020 --model $M --served-model-name qwen3.8-27b" +# --model is the served name (the /tokenize alignment probe posts it as the +# request's model; a checkpoint path 404s there); --tokenizer loads locally. +B="venv/bin/vllm bench serve --host 127.0.0.1 --port 18020 --model qwen3.8-27b --tokenizer $M --served-model-name qwen3.8-27b" metrics() { curl -s http://127.0.0.1:18020/metrics -H "Authorization: Bearer $OPENAI_API_KEY"; } snap() { metrics | grep -E "^vllm:spec_decode_num_(drafts|accepted_tokens)_total" | grep -v created | awk '{print $NF}' | tr "\n" " "; } for i in $(seq 1 $N); do diff --git a/bench/run_benchmarks.sh b/bench/run_benchmarks.sh index 3b3ab1a5..d3502617 100755 --- a/bench/run_benchmarks.sh +++ b/bench/run_benchmarks.sh @@ -22,7 +22,12 @@ export PATH="$REPO/venv/bin:$PATH" export OPENAI_API_KEY=${VLLM_API_KEY:-$(cat "$REPO/api_key.txt" 2>/dev/null)} HOST=${HOST:-127.0.0.1}; PORT=${PORT:-18020} MODEL=${MODEL:-$REPO/models/Qwen3.8-27B-W4A16-AutoRound} -B="venv/bin/vllm bench serve --host $HOST --port $PORT --model $MODEL --served-model-name qwen3.8-27b" +# --model must be the SERVED name, not the checkpoint path: vllm bench serve's +# tokenizer-alignment probe posts it to /tokenize as the request's model, and a +# filesystem path 404s the model check there — the run then logs "WARNING: +# /tokenize unavailable" and silently skips alignment. --tokenizer keeps +# loading the tokenizer from the checkpoint dir, which is what the path is for. +B="venv/bin/vllm bench serve --host $HOST --port $PORT --model qwen3.8-27b --tokenizer $MODEL --served-model-name qwen3.8-27b" OUT=${OUT:-$HERE/results}; mkdir -p "$OUT" curl -sf -o /dev/null http://$HOST:$PORT/health || { echo "no server on $HOST:$PORT"; exit 1; } diff --git a/bench/warmup.sh b/bench/warmup.sh index 3cbddbfd..abcd91e5 100755 --- a/bench/warmup.sh +++ b/bench/warmup.sh @@ -47,7 +47,10 @@ fi # re-globs, so a path with a space or glob character would break, and a # quoted "$B" would not word-split at all. "${BENCH[@]}" expands each element # verbatim. (venv/bin/vllm is relative because we cd "$REPO" above.) -BENCH=(venv/bin/vllm bench serve --host "$HOST" --port "$PORT" --model "$MODEL" --served-model-name qwen3.8-27b) +# --model is the served name (the bench client's /tokenize probe posts it as +# the request's model; a checkpoint path 404s the model check there); the +# checkpoint dir rides --tokenizer, which is all the warmup needs from it. +BENCH=(venv/bin/vllm bench serve --host "$HOST" --port "$PORT" --model qwen3.8-27b --tokenizer "$MODEL" --served-model-name qwen3.8-27b) curl -sf -o /dev/null "http://$HOST:$PORT/health" || { echo "[warmup] no server on $HOST:$PORT" >&2; exit 1; } echo "[warmup] server ready, warming the serving path"