Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion bench/prefill_ab.sh
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,10 @@ OUT="$HERE/results-prefill-ab/$ARM"; mkdir -p "$OUT"
export PATH="$REPO/venv/bin:$PATH"
export OPENAI_API_KEY=${VLLM_API_KEY:-$(cat "$REPO/api_key.txt" 2>/dev/null)}
MODEL=${MODEL:-$REPO/models/Qwen3.8-27B-W4A16-AutoRound-fast}
B="venv/bin/vllm bench serve --host 127.0.0.1 --port $PORT --model $MODEL --served-model-name qwen3.8-27b"
# --model is the served name (the bench client's /tokenize alignment probe
# posts it as the request's model; a checkpoint path 404s there); the
# checkpoint dir rides --tokenizer, which is what actually reads it.
B="venv/bin/vllm bench serve --host 127.0.0.1 --port $PORT --model qwen3.8-27b --tokenizer $MODEL --served-model-name qwen3.8-27b"

# ---- boot -------------------------------------------------------------------
if curl -sf -o /dev/null http://127.0.0.1:$PORT/health; then
Expand Down
4 changes: 3 additions & 1 deletion bench/real_rep.sh
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,9 @@ HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"; REPO="$(dirname "$HERE")";
export PATH="$REPO/venv/bin:$PATH"
export OPENAI_API_KEY=${VLLM_API_KEY:-$(cat "$REPO/api_key.txt" 2>/dev/null)}
M=${MODEL:-$REPO/models/Qwen3.8-27B-W4A16-AutoRound}; TAG=$1; N=${2:-3}; T=${3:-}
B="venv/bin/vllm bench serve --host 127.0.0.1 --port 18020 --model $M --served-model-name qwen3.8-27b"
# --model is the served name (the /tokenize alignment probe posts it as the
# request's model; a checkpoint path 404s there); --tokenizer loads locally.
B="venv/bin/vllm bench serve --host 127.0.0.1 --port 18020 --model qwen3.8-27b --tokenizer $M --served-model-name qwen3.8-27b"
metrics() { curl -s http://127.0.0.1:18020/metrics -H "Authorization: Bearer $OPENAI_API_KEY"; }
snap() { metrics | grep -E "^vllm:spec_decode_num_(drafts|accepted_tokens)_total" | grep -v created | awk '{print $NF}' | tr "\n" " "; }
for i in $(seq 1 $N); do
Expand Down
7 changes: 6 additions & 1 deletion bench/run_benchmarks.sh
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,12 @@ export PATH="$REPO/venv/bin:$PATH"
export OPENAI_API_KEY=${VLLM_API_KEY:-$(cat "$REPO/api_key.txt" 2>/dev/null)}
HOST=${HOST:-127.0.0.1}; PORT=${PORT:-18020}
MODEL=${MODEL:-$REPO/models/Qwen3.8-27B-W4A16-AutoRound}
B="venv/bin/vllm bench serve --host $HOST --port $PORT --model $MODEL --served-model-name qwen3.8-27b"
# --model must be the SERVED name, not the checkpoint path: vllm bench serve's
# tokenizer-alignment probe posts it to /tokenize as the request's model, and a
# filesystem path 404s the model check there — the run then logs "WARNING:
# /tokenize unavailable" and silently skips alignment. --tokenizer keeps
# loading the tokenizer from the checkpoint dir, which is what the path is for.
B="venv/bin/vllm bench serve --host $HOST --port $PORT --model qwen3.8-27b --tokenizer $MODEL --served-model-name qwen3.8-27b"
OUT=${OUT:-$HERE/results}; mkdir -p "$OUT"

curl -sf -o /dev/null http://$HOST:$PORT/health || { echo "no server on $HOST:$PORT"; exit 1; }
Expand Down
5 changes: 4 additions & 1 deletion bench/warmup.sh
Original file line number Diff line number Diff line change
Expand Up @@ -47,7 +47,10 @@ fi
# re-globs, so a path with a space or glob character would break, and a
# quoted "$B" would not word-split at all. "${BENCH[@]}" expands each element
# verbatim. (venv/bin/vllm is relative because we cd "$REPO" above.)
BENCH=(venv/bin/vllm bench serve --host "$HOST" --port "$PORT" --model "$MODEL" --served-model-name qwen3.8-27b)
# --model is the served name (the bench client's /tokenize probe posts it as
# the request's model; a checkpoint path 404s the model check there); the
# checkpoint dir rides --tokenizer, which is all the warmup needs from it.
BENCH=(venv/bin/vllm bench serve --host "$HOST" --port "$PORT" --model qwen3.8-27b --tokenizer "$MODEL" --served-model-name qwen3.8-27b)

curl -sf -o /dev/null "http://$HOST:$PORT/health" || { echo "[warmup] no server on $HOST:$PORT" >&2; exit 1; }
echo "[warmup] server ready, warming the serving path"
Expand Down
Loading