From 4ac4d21acc32e99965d7b690ddf5b13ebc94ce9b Mon Sep 17 00:00:00 2001 From: Tyrone <71038642+TyroneNel@users.noreply.github.com> Date: Mon, 21 Sep 2026 21:14:51 +0000 Subject: [PATCH 1/4] =?UTF-8?q?patches:=20bench-probe-errors=20=E2=80=94?= =?UTF-8?q?=20the=20alignment=20probe=20classifies=20failures=20and=20send?= =?UTF-8?q?s=20the=20key?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit vllm bench serve's tokenizer-alignment probe collapsed every failure into one wrong message ("/tokenize unavailable") and sent no Authorization header, so a keyed server 401'd it even with a valid model name. The patch sends the same Bearer the benchmark requests carry and names the real cause: 404 (no route, or the model name was rejected — /v1/models lists the served names), 401 (key required), unreachable, timeout, or the raw exception. Independent of every other patch in the series (benchmarks/serve.py is untouched by them), so it rides at the end of patches/series. Cut from the extended cpuchip/vllm qwen38/0.28 branch, topic commit [qwen38] bench-probe-errors; kind: fix, retires when upstream takes it. --- PATCHES.md | 1 + patches/bench-probe-errors.patch | 106 +++++++++++++++++++++++++++++++ patches/series | 1 + 3 files changed, 108 insertions(+) create mode 100644 patches/bench-probe-errors.patch diff --git a/PATCHES.md b/PATCHES.md index 7cf6656e..308343dc 100644 --- a/PATCHES.md +++ b/PATCHES.md @@ -19,6 +19,7 @@ build by name instead of landing by guess. Regenerate a file with `bash scripts/ | patch | kind | what | upstream | cut against | retires when | |---|---|---|---|---|---| +| bench-probe-errors | fix | `vllm bench serve`'s /tokenize alignment probe sends the API key (Bearer from OPENAI_API_KEY, --header wins) and classifies its failure (404 route-or-name vs 401 vs unreachable vs timeout) instead of one "endpoint unavailable" line for every cause | none yet | 0.28.0 | upstream PR | | dflash2-backport | backport, RETIRED | DFlash2 speculator on 0.27.1 | vllm #52816 (in 0.28.0) | 0.27.1 | done; kept for history, skipped by the Dockerfile | | dflash2-lookup-drafting | feature | lookup-augmented drafting for DFlash2 (n-gram search over the context); registers its `VLLM_DFLASH2_LOOKUP*`, `VLLM_DFLASH2_GRAPH_BOTH`, `VLLM_DFLASH2_DRAFT_TOPK_TOPP` knobs | none | 0.28.0 | upstreamed | | dflash2-ngram-chains | feature | quantized candidate chains for the drafter; `propose` override; registers `VLLM_DFLASH2_CHAIN*` | none | 0.28.0 | upstreamed | diff --git a/patches/bench-probe-errors.patch b/patches/bench-probe-errors.patch new file mode 100644 index 00000000..f8d7f708 --- /dev/null +++ b/patches/bench-probe-errors.patch @@ -0,0 +1,106 @@ +Name the alignment probe's real failure instead of one wrong message. + +`vllm bench serve` re-aligns random/prefix_repetition prompts to the server's +tokenizer through /tokenize and /detokenize. Every failure of that probe -- +a 404 model check (it posts `--model` verbatim as the request's model, so a +checkpoint path 404s), a 401 on a keyed server (it sent no Authorization +header at all), a missing route, a timeout, an unreachable host -- collapsed +into the same "WARNING: /tokenize unavailable, skipping alignment." line, +which sends the operator hunting a broken server when the server is fine. + +The probe now: + - sends the same Authorization the benchmark requests carry (Bearer from + OPENAI_API_KEY, unless --header already supplied one); + - classifies the first failure: 401 names the key, 404 names both possible + causes (no /tokenize route at this base_url, or the model name was + rejected -- GET /v1/models lists the names the server serves), connection + errors name the host, timeouts say so, and anything else prints the real + exception. + +Alignment behaviour is unchanged: it still runs only for str prompts, still +skips on failure rather than failing the run. + +--- exported from cpuchip/vllm 813321b (bench-probe-errors); regenerate with scripts/export-patch.sh, do not edit --- + +diff --git a/benchmarks/serve.py b/benchmarks/serve.py +index 9b99368..e5de133 100644 +--- a/benchmarks/serve.py ++++ b/benchmarks/serve.py +@@ -76,6 +76,7 @@ + model_id: str, + input_requests: list[SampleRequest], + ssl_context: ssl.SSLContext | bool | None = None, ++ headers: dict[str, str] | None = None, + ) -> list[SampleRequest]: + """Re-align prompts if local/server tokenizers disagree.""" + if not input_requests or not isinstance(input_requests[0].prompt, str): +@@ -83,6 +84,11 @@ + + tok_url = f"{base_url}/tokenize" + detok_url = f"{base_url}/detokenize" ++ # The benchmark requests carry the key from OPENAI_API_KEY; this probe ++ # must too, or a server bound with --api-key rejects it with a 401. ++ probe_headers = dict(headers or {}) ++ if api_key := os.environ.get("OPENAI_API_KEY"): ++ probe_headers.setdefault("Authorization", f"Bearer {api_key}") + connector = aiohttp.TCPConnector(ssl=ssl_context) + + async with aiohttp.ClientSession(connector=connector) as session: +@@ -98,6 +104,7 @@ + "prompt": prompt, + "add_special_tokens": False, + }, ++ headers=probe_headers, + ) as r, + ): + r.raise_for_status() +@@ -107,7 +114,9 @@ + async with ( + sem, + session.post( +- detok_url, json={"model": model_id, "tokens": tokens} ++ detok_url, ++ json={"model": model_id, "tokens": tokens}, ++ headers=probe_headers, + ) as r, + ): + r.raise_for_status() +@@ -115,8 +124,27 @@ + + try: + first_tokens = await _tokenize(input_requests[0].prompt) +- except Exception: +- print("WARNING: /tokenize unavailable, skipping alignment.") ++ except asyncio.TimeoutError: ++ print("WARNING: /tokenize probe timed out, skipping alignment.") ++ return input_requests ++ except aiohttp.ClientConnectionError as e: ++ print(f"WARNING: {base_url} unreachable ({e!r}), skipping alignment.") ++ return input_requests ++ except aiohttp.ClientResponseError as e: ++ if e.status == 401: ++ hint = "401 Unauthorized: the server requires an API key" ++ elif e.status == 404: ++ hint = ( ++ "404 Not Found: either this server has no /tokenize route," ++ f" or it does not serve a model named `{model_id}`" ++ " (its served names are listed by GET /v1/models)" ++ ) ++ else: ++ hint = f"HTTP {e.status}" ++ print(f"WARNING: /tokenize unavailable ({hint}), skipping alignment.") ++ return input_requests ++ except Exception as e: ++ print(f"WARNING: /tokenize probe failed ({e!r}), skipping alignment.") + return input_requests + + expected = input_requests[0].prompt_len +@@ -2124,7 +2152,7 @@ + + if args.dataset_name in ("random", "prefix_repetition"): + input_requests = await _align_prompts_to_server_tokenizer( +- base_url, model_id, input_requests, ssl_context ++ base_url, model_id, input_requests, ssl_context, headers=headers + ) + + goodput_config_dict = check_goodput_args(args) diff --git a/patches/series b/patches/series index cb062844..f1b29b41 100644 --- a/patches/series +++ b/patches/series @@ -51,3 +51,4 @@ engine-stall-sentinel.patch sse-keep-alive.patch int4-mq3d-envs.patch triton-spec-attn-fp8-kv.patch +bench-probe-errors.patch From 5c8755d7dddc8c254cc250a2723f12611f6a803f Mon Sep 17 00:00:00 2001 From: Tyrone <71038642+TyroneNel@users.noreply.github.com> Date: Tue, 22 Sep 2026 00:26:52 +0200 Subject: [PATCH 2/4] PATCHES.md: link the upstream vLLM PR for bench-probe-errors (vllm-project/vllm#58024) --- PATCHES.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/PATCHES.md b/PATCHES.md index 308343dc..447995fe 100644 --- a/PATCHES.md +++ b/PATCHES.md @@ -19,7 +19,7 @@ build by name instead of landing by guess. Regenerate a file with `bash scripts/ | patch | kind | what | upstream | cut against | retires when | |---|---|---|---|---|---| -| bench-probe-errors | fix | `vllm bench serve`'s /tokenize alignment probe sends the API key (Bearer from OPENAI_API_KEY, --header wins) and classifies its failure (404 route-or-name vs 401 vs unreachable vs timeout) instead of one "endpoint unavailable" line for every cause | none yet | 0.28.0 | upstream PR | +| bench-probe-errors | fix | `vllm bench serve`'s /tokenize alignment probe sends the API key (Bearer from OPENAI_API_KEY, --header wins) and classifies its failure (404 route-or-name vs 401 vs unreachable vs timeout) instead of one "endpoint unavailable" line for every cause | vllm #58024 | 0.28.0 | upstream PR | | dflash2-backport | backport, RETIRED | DFlash2 speculator on 0.27.1 | vllm #52816 (in 0.28.0) | 0.27.1 | done; kept for history, skipped by the Dockerfile | | dflash2-lookup-drafting | feature | lookup-augmented drafting for DFlash2 (n-gram search over the context); registers its `VLLM_DFLASH2_LOOKUP*`, `VLLM_DFLASH2_GRAPH_BOTH`, `VLLM_DFLASH2_DRAFT_TOPK_TOPP` knobs | none | 0.28.0 | upstreamed | | dflash2-ngram-chains | feature | quantized candidate chains for the drafter; `propose` override; registers `VLLM_DFLASH2_CHAIN*` | none | 0.28.0 | upstreamed | From ba38d5c308affa0f5962fa6ffffbfdcdfa6b9766 Mon Sep 17 00:00:00 2001 From: Tyrone <71038642+TyroneNel@users.noreply.github.com> Date: Tue, 22 Sep 2026 13:19:36 +0200 Subject: [PATCH 3/4] =?UTF-8?q?patches:=20bench-probe-errors=20=E2=80=94?= =?UTF-8?q?=20the=20/metrics=20scrapes=20send=20the=20key=20too?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit fetch_spec_decode_metrics and fetch_diffusion_metrics GET /metrics with no headers and return None on any non-200, so on a server bound with --api-key the 401 is indistinguishable from 'speculative decoding is off' and the benchmark's whole spec-decode block (acceptance length, accepted/drafted, per-position acceptance) silently disappears from every run. Both fetchers now take the headers the benchmark already built; the four call sites in benchmark() pass extra_headers, the same dict the request path uses. Unkeyed servers and the None-on-missing-metrics contract are unchanged. Observed on hyperqwen d6e094a3 with a 48-char VLLM_API_KEY: GET /metrics keyless 401, with the key 200, /health 200 keyless; the bench client's 401s share an ephemeral port with its own POST /v1/completions, which is what identifies it (rather than run_benchmarks.sh's curl) as the caller. --- PATCHES.md | 2 +- patches/bench-probe-errors.patch | 106 ++++++++++++++++++++++++++++--- 2 files changed, 99 insertions(+), 9 deletions(-) diff --git a/PATCHES.md b/PATCHES.md index 447995fe..be8994ad 100644 --- a/PATCHES.md +++ b/PATCHES.md @@ -19,7 +19,7 @@ build by name instead of landing by guess. Regenerate a file with `bash scripts/ | patch | kind | what | upstream | cut against | retires when | |---|---|---|---|---|---| -| bench-probe-errors | fix | `vllm bench serve`'s /tokenize alignment probe sends the API key (Bearer from OPENAI_API_KEY, --header wins) and classifies its failure (404 route-or-name vs 401 vs unreachable vs timeout) instead of one "endpoint unavailable" line for every cause | vllm #58024 | 0.28.0 | upstream PR | +| bench-probe-errors | fix | `vllm bench serve`'s /tokenize alignment probe sends the API key (Bearer from OPENAI_API_KEY, --header wins) and classifies its failure (404 route-or-name vs 401 vs unreachable vs timeout) instead of one "endpoint unavailable" line for every cause; the /metrics scrapes (`fetch_spec_decode_metrics`, `fetch_diffusion_metrics`) send the benchmark's headers too, so a keyed server no longer reports the spec-decode block as absent | vllm #58024 | 0.28.0 | upstream PR | | dflash2-backport | backport, RETIRED | DFlash2 speculator on 0.27.1 | vllm #52816 (in 0.28.0) | 0.27.1 | done; kept for history, skipped by the Dockerfile | | dflash2-lookup-drafting | feature | lookup-augmented drafting for DFlash2 (n-gram search over the context); registers its `VLLM_DFLASH2_LOOKUP*`, `VLLM_DFLASH2_GRAPH_BOTH`, `VLLM_DFLASH2_DRAFT_TOPK_TOPP` knobs | none | 0.28.0 | upstreamed | | dflash2-ngram-chains | feature | quantized candidate chains for the drafter; `propose` override; registers `VLLM_DFLASH2_CHAIN*` | none | 0.28.0 | upstreamed | diff --git a/patches/bench-probe-errors.patch b/patches/bench-probe-errors.patch index f8d7f708..dd23ce40 100644 --- a/patches/bench-probe-errors.patch +++ b/patches/bench-probe-errors.patch @@ -1,4 +1,5 @@ -Name the alignment probe's real failure instead of one wrong message. +Name the alignment probe's real failure instead of one wrong message, and +send the key on the metrics scrapes too. `vllm bench serve` re-aligns random/prefix_repetition prompts to the server's tokenizer through /tokenize and /detokenize. Every failure of that probe -- @@ -17,16 +18,28 @@ The probe now: errors name the host, timeouts say so, and anything else prints the real exception. +`fetch_spec_decode_metrics` and `fetch_diffusion_metrics` had the same +omission with a quieter symptom: both GET /metrics with no headers, and both +return None on any non-200. Against a server bound with --api-key that 401 is +indistinguishable from "speculative decoding is off", so the benchmark's +spec-decode block -- acceptance length, accepted/drafted counts, per-position +acceptance -- silently disappears from every run on a keyed server, and the +operator reads acceptance out of the engine log instead. Both fetchers now +take the headers the benchmark already built and pass them to the GET; the +four call sites in `benchmark()` hand over `extra_headers`, which is the same +dict the request path uses. Behaviour against an unkeyed server, and the +None-on-missing-metrics contract, are unchanged. + Alignment behaviour is unchanged: it still runs only for str prompts, still skips on failure rather than failing the run. --- exported from cpuchip/vllm 813321b (bench-probe-errors); regenerate with scripts/export-patch.sh, do not edit --- diff --git a/benchmarks/serve.py b/benchmarks/serve.py -index 9b99368..e5de133 100644 +index 9b99368..f284f4a 100644 --- a/benchmarks/serve.py +++ b/benchmarks/serve.py -@@ -76,6 +76,7 @@ +@@ -76,6 +76,7 @@ async def _align_prompts_to_server_tokenizer( model_id: str, input_requests: list[SampleRequest], ssl_context: ssl.SSLContext | bool | None = None, @@ -34,7 +47,7 @@ index 9b99368..e5de133 100644 ) -> list[SampleRequest]: """Re-align prompts if local/server tokenizers disagree.""" if not input_requests or not isinstance(input_requests[0].prompt, str): -@@ -83,6 +84,11 @@ +@@ -83,6 +84,11 @@ async def _align_prompts_to_server_tokenizer( tok_url = f"{base_url}/tokenize" detok_url = f"{base_url}/detokenize" @@ -46,7 +59,7 @@ index 9b99368..e5de133 100644 connector = aiohttp.TCPConnector(ssl=ssl_context) async with aiohttp.ClientSession(connector=connector) as session: -@@ -98,6 +104,7 @@ +@@ -98,6 +104,7 @@ async def _align_prompts_to_server_tokenizer( "prompt": prompt, "add_special_tokens": False, }, @@ -54,7 +67,7 @@ index 9b99368..e5de133 100644 ) as r, ): r.raise_for_status() -@@ -107,7 +114,9 @@ +@@ -107,7 +114,9 @@ async def _align_prompts_to_server_tokenizer( async with ( sem, session.post( @@ -65,7 +78,7 @@ index 9b99368..e5de133 100644 ) as r, ): r.raise_for_status() -@@ -115,8 +124,27 @@ +@@ -115,8 +124,27 @@ async def _align_prompts_to_server_tokenizer( try: first_tokens = await _tokenize(input_requests[0].prompt) @@ -95,7 +108,84 @@ index 9b99368..e5de133 100644 return input_requests expected = input_requests[0].prompt_len -@@ -2124,7 +2152,7 @@ +@@ -187,7 +215,9 @@ class SpecDecodeMetrics: + + + async def fetch_spec_decode_metrics( +- base_url: str, session: aiohttp.ClientSession ++ base_url: str, ++ session: aiohttp.ClientSession, ++ headers: dict[str, str] | None = None, + ) -> SpecDecodeMetrics | None: + """Fetch speculative decoding metrics from the server's Prometheus endpoint. + +@@ -195,7 +225,7 @@ async def fetch_spec_decode_metrics( + """ + metrics_url = f"{base_url}/metrics" + try: +- async with session.get(metrics_url) as response: ++ async with session.get(metrics_url, headers=headers) as response: + if response.status != 200: + return None + text = await response.text() +@@ -260,7 +290,9 @@ class DiffusionMetrics: + + + async def fetch_diffusion_metrics( +- base_url: str, session: aiohttp.ClientSession ++ base_url: str, ++ session: aiohttp.ClientSession, ++ headers: dict[str, str] | None = None, + ) -> DiffusionMetrics | None: + """Fetch diffusion decoding metrics from the server's Prometheus endpoint. + +@@ -269,7 +301,7 @@ async def fetch_diffusion_metrics( + """ + metrics_url = f"{base_url}/metrics" + try: +- async with session.get(metrics_url) as response: ++ async with session.get(metrics_url, headers=headers) as response: + if response.status != 200: + return None + text = await response.text() +@@ -955,8 +987,12 @@ async def benchmark( + else: + print("Self timing is set, using the timestamps from the trace file.") + +- spec_decode_metrics_before = await fetch_spec_decode_metrics(base_url, session) +- diffusion_metrics_before = await fetch_diffusion_metrics(base_url, session) ++ spec_decode_metrics_before = await fetch_spec_decode_metrics( ++ base_url, session, headers=extra_headers ++ ) ++ diffusion_metrics_before = await fetch_diffusion_metrics( ++ base_url, session, headers=extra_headers ++ ) + + pbar = None if disable_tqdm else tqdm(total=len(input_requests)) + +@@ -1076,7 +1112,9 @@ async def benchmark( + + benchmark_duration = time.perf_counter() - benchmark_start_time + +- spec_decode_metrics_after = await fetch_spec_decode_metrics(base_url, session) ++ spec_decode_metrics_after = await fetch_spec_decode_metrics( ++ base_url, session, headers=extra_headers ++ ) + spec_decode_stats: dict[str, Any] | None = None + if spec_decode_metrics_before is not None and spec_decode_metrics_after is not None: + delta_drafts = ( +@@ -1118,7 +1156,9 @@ async def benchmark( + "per_position_acceptance_rates": per_pos_rates, + } + +- diffusion_metrics_after = await fetch_diffusion_metrics(base_url, session) ++ diffusion_metrics_after = await fetch_diffusion_metrics( ++ base_url, session, headers=extra_headers ++ ) + diffusion_stats: dict[str, Any] | None = None + if diffusion_metrics_before is not None and diffusion_metrics_after is not None: + delta_steps = ( +@@ -2124,7 +2164,7 @@ async def main_async(args: argparse.Namespace) -> dict[str, Any]: if args.dataset_name in ("random", "prefix_repetition"): input_requests = await _align_prompts_to_server_tokenizer( From 84c0b5221c8d07d2fd829e4cb9fc05d6460fccb7 Mon Sep 17 00:00:00 2001 From: Tyrone <71038642+TyroneNel@users.noreply.github.com> Date: Tue, 22 Sep 2026 18:28:34 +0200 Subject: [PATCH 4/4] =?UTF-8?q?patches:=20bench-probe-errors=20=E2=80=94?= =?UTF-8?q?=20the=20metrics=20scrapes=20build=20their=20own=20key=20header?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Threading extra_headers into fetch_spec_decode_metrics and fetch_diffusion_metrics was not enough. That dict only ever carries the --header pairs, so on a server bound with --api-key and no --header it is empty, and all four /metrics scrapes still came back 401 -- which both fetchers return as None, indistinguishable from 'speculative decoding is off'. The spec-decode block then vanished from every run on a keyed server, which is the exact symptom the patch set out to fix. Both fetchers now build their scrape headers the way the /tokenize probe does: start from whatever the caller passed, then add the Bearer from OPENAI_API_KEY unless one is already set. Observed on a keyed server: repeated 'GET /metrics 401 Unauthorized' from a single long-lived bench-client session, while the harness's own keyed curl scrapes on neighbouring ports returned 200. --- patches/bench-probe-errors.patch | 53 ++++++++++++++++++++------------ 1 file changed, 34 insertions(+), 19 deletions(-) diff --git a/patches/bench-probe-errors.patch b/patches/bench-probe-errors.patch index dd23ce40..f40ab367 100644 --- a/patches/bench-probe-errors.patch +++ b/patches/bench-probe-errors.patch @@ -25,10 +25,12 @@ indistinguishable from "speculative decoding is off", so the benchmark's spec-decode block -- acceptance length, accepted/drafted counts, per-position acceptance -- silently disappears from every run on a keyed server, and the operator reads acceptance out of the engine log instead. Both fetchers now -take the headers the benchmark already built and pass them to the GET; the -four call sites in `benchmark()` hand over `extra_headers`, which is the same -dict the request path uses. Behaviour against an unkeyed server, and the -None-on-missing-metrics contract, are unchanged. +build their own scrape headers the way the probe does: start from whatever +the caller passed, then add the Bearer from OPENAI_API_KEY unless one is +already set. Threading `extra_headers` down to them is not enough on its own +-- that dict only ever carries `--header` pairs, so on a keyed server with no +`--header` it is empty and every scrape still 401s. Behaviour against an +unkeyed server, and the None-on-missing-metrics contract, are unchanged. Alignment behaviour is unchanged: it still runs only for str prompts, still skips on failure rather than failing the run. @@ -39,7 +41,7 @@ diff --git a/benchmarks/serve.py b/benchmarks/serve.py index 9b99368..f284f4a 100644 --- a/benchmarks/serve.py +++ b/benchmarks/serve.py -@@ -76,6 +76,7 @@ async def _align_prompts_to_server_tokenizer( +@@ -76,6 +76,7 @@ model_id: str, input_requests: list[SampleRequest], ssl_context: ssl.SSLContext | bool | None = None, @@ -47,7 +49,7 @@ index 9b99368..f284f4a 100644 ) -> list[SampleRequest]: """Re-align prompts if local/server tokenizers disagree.""" if not input_requests or not isinstance(input_requests[0].prompt, str): -@@ -83,6 +84,11 @@ async def _align_prompts_to_server_tokenizer( +@@ -83,6 +84,11 @@ tok_url = f"{base_url}/tokenize" detok_url = f"{base_url}/detokenize" @@ -59,7 +61,7 @@ index 9b99368..f284f4a 100644 connector = aiohttp.TCPConnector(ssl=ssl_context) async with aiohttp.ClientSession(connector=connector) as session: -@@ -98,6 +104,7 @@ async def _align_prompts_to_server_tokenizer( +@@ -98,6 +104,7 @@ "prompt": prompt, "add_special_tokens": False, }, @@ -67,7 +69,7 @@ index 9b99368..f284f4a 100644 ) as r, ): r.raise_for_status() -@@ -107,7 +114,9 @@ async def _align_prompts_to_server_tokenizer( +@@ -107,7 +114,9 @@ async with ( sem, session.post( @@ -78,7 +80,7 @@ index 9b99368..f284f4a 100644 ) as r, ): r.raise_for_status() -@@ -115,8 +124,27 @@ async def _align_prompts_to_server_tokenizer( +@@ -115,8 +124,27 @@ try: first_tokens = await _tokenize(input_requests[0].prompt) @@ -108,7 +110,7 @@ index 9b99368..f284f4a 100644 return input_requests expected = input_requests[0].prompt_len -@@ -187,7 +215,9 @@ class SpecDecodeMetrics: +@@ -187,15 +215,23 @@ async def fetch_spec_decode_metrics( @@ -119,16 +121,22 @@ index 9b99368..f284f4a 100644 ) -> SpecDecodeMetrics | None: """Fetch speculative decoding metrics from the server's Prometheus endpoint. -@@ -195,7 +225,7 @@ async def fetch_spec_decode_metrics( + Returns None if speculative decoding is not enabled or metrics are not available. """ metrics_url = f"{base_url}/metrics" ++ # Same requirement as the /tokenize probe above: a server bound with ++ # --api-key rejects an unauthenticated scrape, and the caller's ++ # extra_headers only carries --header pairs, never the key. ++ scrape_headers = dict(headers or {}) ++ if api_key := os.environ.get("OPENAI_API_KEY"): ++ scrape_headers.setdefault("Authorization", f"Bearer {api_key}") try: - async with session.get(metrics_url) as response: -+ async with session.get(metrics_url, headers=headers) as response: ++ async with session.get(metrics_url, headers=scrape_headers) as response: if response.status != 200: return None text = await response.text() -@@ -260,7 +290,9 @@ class DiffusionMetrics: +@@ -260,7 +296,9 @@ async def fetch_diffusion_metrics( @@ -139,16 +147,23 @@ index 9b99368..f284f4a 100644 ) -> DiffusionMetrics | None: """Fetch diffusion decoding metrics from the server's Prometheus endpoint. -@@ -269,7 +301,7 @@ async def fetch_diffusion_metrics( +@@ -268,8 +306,14 @@ + available. """ metrics_url = f"{base_url}/metrics" ++ # Same requirement as the /tokenize probe above: a server bound with ++ # --api-key rejects an unauthenticated scrape, and the caller's ++ # extra_headers only carries --header pairs, never the key. ++ scrape_headers = dict(headers or {}) ++ if api_key := os.environ.get("OPENAI_API_KEY"): ++ scrape_headers.setdefault("Authorization", f"Bearer {api_key}") try: - async with session.get(metrics_url) as response: -+ async with session.get(metrics_url, headers=headers) as response: ++ async with session.get(metrics_url, headers=scrape_headers) as response: if response.status != 200: return None text = await response.text() -@@ -955,8 +987,12 @@ async def benchmark( +@@ -955,8 +999,12 @@ else: print("Self timing is set, using the timestamps from the trace file.") @@ -163,7 +178,7 @@ index 9b99368..f284f4a 100644 pbar = None if disable_tqdm else tqdm(total=len(input_requests)) -@@ -1076,7 +1112,9 @@ async def benchmark( +@@ -1076,7 +1124,9 @@ benchmark_duration = time.perf_counter() - benchmark_start_time @@ -174,7 +189,7 @@ index 9b99368..f284f4a 100644 spec_decode_stats: dict[str, Any] | None = None if spec_decode_metrics_before is not None and spec_decode_metrics_after is not None: delta_drafts = ( -@@ -1118,7 +1156,9 @@ async def benchmark( +@@ -1118,7 +1168,9 @@ "per_position_acceptance_rates": per_pos_rates, } @@ -185,7 +200,7 @@ index 9b99368..f284f4a 100644 diffusion_stats: dict[str, Any] | None = None if diffusion_metrics_before is not None and diffusion_metrics_after is not None: delta_steps = ( -@@ -2124,7 +2164,7 @@ async def main_async(args: argparse.Namespace) -> dict[str, Any]: +@@ -2124,7 +2176,7 @@ if args.dataset_name in ("random", "prefix_repetition"): input_requests = await _align_prompts_to_server_tokenizer(