From dee6f1f73a04a9cd4d0f74b41a1fc96b0737559b Mon Sep 17 00:00:00 2001 From: Kaushalya Date: Sun, 16 Aug 2026 22:22:48 +0900 Subject: [PATCH 1/5] Add spec-decode recording rules and Grafana dashboard - prometheus-rules.yaml: new vllm-spec-decode group with five 1m-rate recording rules (acceptance rate, mean accepted length, draft/accepted token rates, per-position acceptance) following the existing {job="vllm"} + 5s interval conventions - grafana/dashboards/vllm-spec-decode.json: new provisioned "vLLM Speculative Decoding" dashboard with five panels querying the recorded metrics - docs/METRICS.md: document the dashboard and recording rules; keep the raw-counter PromQL as the equivalent reference --- docs/METRICS.md | 83 +++++-- grafana/dashboards/vllm-spec-decode.json | 290 +++++++++++++++++++++++ prometheus-rules.yaml | 48 ++++ 3 files changed, 405 insertions(+), 16 deletions(-) create mode 100644 grafana/dashboards/vllm-spec-decode.json diff --git a/docs/METRICS.md b/docs/METRICS.md index 5162479b..13e7535b 100644 --- a/docs/METRICS.md +++ b/docs/METRICS.md @@ -12,11 +12,15 @@ The repository includes a ready-to-run monitoring stack: - `prometheus.yaml` scrapes the vLLM server from the Docker host every five seconds. - `prometheus-rules.yaml` records one-minute averages for prompt, generation, - and total token throughput. + and total token throughput, plus speculative-decoding metrics (acceptance + rate, mean accepted length, draft and accepted token rates, and + per-position acceptance). - `grafana/provisioning/` configures Prometheus as Grafana's default data source and loads dashboards from disk. - `grafana/dashboards/vllm-throughput.json` defines the default **vLLM Throughput** dashboard. +- `grafana/dashboards/vllm-spec-decode.json` defines the **vLLM Speculative + Decoding** dashboard. The default configuration assumes vLLM is listening on port `8000` on the Docker host. Confirm the endpoint first: @@ -88,16 +92,50 @@ imported into Grafana: ## Speculative decoding and MTP panels -The general vLLM dashboard may not include speculative-decoding panels. Add -Grafana panels with the following PromQL queries. +The provisioned **vLLM Speculative Decoding** dashboard +(`grafana/dashboards/vllm-spec-decode.json`) covers these metrics out of the +box. It appears in the **vLLM** folder and queries the `vllm:spec_decode_*` +recording rules from the `vllm-spec-decode` group in `prometheus-rules.yaml`: + +- **Draft acceptance rate** — fraction of draft tokens accepted + (`vllm:spec_decode_acceptance_rate:rate1m`) +- **Mean accepted length** — tokens emitted per verification step, including + the bonus token (`vllm:spec_decode_mean_accepted_length:rate1m`); with N + speculative tokens it ranges from 1.0 to N+1 +- **Draft tokens / second** (`vllm:spec_decode_draft_tokens_per_second:rate1m`) +- **Draft vs accepted tokens / second** — the gap between the two series is + wasted draft compute +- **Acceptance rate by draft position** — per-position quality of the MTP + cascade (`vllm:spec_decode_acceptance_rate_by_pos:rate1m`) + +The rules evaluate to no value while vLLM is idle, so the panels show **No +value** until requests are being generated; this avoids a misleading 100% +acceptance rate at idle. + +If you add or replace dashboard files on disk, restart the Grafana container +for the file provider to reload them: + +```bash +docker compose -f docker-compose.metrics.yaml restart grafana +``` + +If you prefer to build your own panels directly on the raw counters (for +example with a longer rate window), the following PromQL queries are +equivalent to the recording rules: ### Draft-token acceptance rate ```promql -100 * -sum(rate(vllm:spec_decode_num_accepted_tokens_total[5m])) -/ -sum(rate(vllm:spec_decode_num_draft_tokens_total[5m])) +( + 100 * + sum(rate(vllm:spec_decode_num_accepted_tokens_total[5m])) + / + sum(rate(vllm:spec_decode_num_draft_tokens_total[5m])) +) +and on() +( + sum(rate(vllm:spec_decode_num_draft_tokens_total[5m])) > 0 +) ``` Use a Gauge or Time series visualization with the unit set to percent. @@ -108,10 +146,16 @@ This convention includes the target/bonus token emitted by a verification step: ```promql -1 + -sum(rate(vllm:spec_decode_num_accepted_tokens_total[5m])) -/ -sum(rate(vllm:spec_decode_num_drafts_total[5m])) +( + 1 + + sum(rate(vllm:spec_decode_num_accepted_tokens_total[5m])) + / + sum(rate(vllm:spec_decode_num_drafts_total[5m])) +) +and on() +( + sum(rate(vllm:spec_decode_num_drafts_total[5m])) > 0 +) ``` ### Draft and accepted tokens per second @@ -131,13 +175,20 @@ sum(rate(vllm:spec_decode_num_accepted_tokens_total[5m])) ### Acceptance rate by draft position ```promql -100 * -sum by (position) ( - rate(vllm:spec_decode_num_accepted_tokens_per_pos_total[5m]) +( + 100 * + sum by (position) ( + rate(vllm:spec_decode_num_accepted_tokens_per_pos_total[5m]) + ) + / + scalar( + sum(rate(vllm:spec_decode_num_drafts_total[5m])) + ) ) -/ -scalar( +and on() +( sum(rate(vllm:spec_decode_num_drafts_total[5m])) + > 0 ) ``` diff --git a/grafana/dashboards/vllm-spec-decode.json b/grafana/dashboards/vllm-spec-decode.json new file mode 100644 index 00000000..b50734c4 --- /dev/null +++ b/grafana/dashboards/vllm-spec-decode.json @@ -0,0 +1,290 @@ +{ + "annotations": { + "list": [] + }, + "editable": true, + "fiscalYearStartMonth": 0, + "graphTooltip": 1, + "id": null, + "links": [], + "liveNow": false, + "panels": [ + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "fieldConfig": { + "defaults": { + "decimals": 1, + "max": 1, + "min": 0, + "unit": "percentunit" + }, + "overrides": [] + }, + "gridPos": { + "h": 6, + "w": 8, + "x": 0, + "y": 0 + }, + "id": 1, + "options": { + "colorMode": "value", + "graphMode": "area", + "justifyMode": "auto", + "orientation": "auto", + "reduceOptions": { + "calcs": [ + "lastNotNull" + ], + "fields": "", + "values": false + }, + "showPercentChange": false, + "textMode": "auto", + "wideLayout": true + }, + "pluginVersion": "11.0.0", + "targets": [ + { + "editorMode": "code", + "expr": "vllm:spec_decode_acceptance_rate:rate1m", + "legendFormat": "Accepted", + "range": true, + "refId": "A" + } + ], + "title": "Draft acceptance rate (1-minute average)", + "type": "stat" + }, + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "fieldConfig": { + "defaults": { + "decimals": 2, + "min": 1, + "unit": "short" + }, + "overrides": [] + }, + "gridPos": { + "h": 6, + "w": 8, + "x": 8, + "y": 0 + }, + "id": 2, + "options": { + "colorMode": "value", + "graphMode": "area", + "justifyMode": "auto", + "orientation": "auto", + "reduceOptions": { + "calcs": [ + "lastNotNull" + ], + "fields": "", + "values": false + }, + "showPercentChange": false, + "textMode": "auto", + "wideLayout": true + }, + "pluginVersion": "11.0.0", + "targets": [ + { + "editorMode": "code", + "expr": "vllm:spec_decode_mean_accepted_length:rate1m", + "legendFormat": "Tokens / step", + "range": true, + "refId": "A" + } + ], + "title": "Mean accepted length (incl. bonus token)", + "type": "stat" + }, + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "fieldConfig": { + "defaults": { + "decimals": 1, + "min": 0, + "unit": "suffix: tokens/s" + }, + "overrides": [] + }, + "gridPos": { + "h": 6, + "w": 8, + "x": 16, + "y": 0 + }, + "id": 3, + "options": { + "colorMode": "value", + "graphMode": "area", + "justifyMode": "auto", + "orientation": "auto", + "reduceOptions": { + "calcs": [ + "lastNotNull" + ], + "fields": "", + "values": false + }, + "showPercentChange": false, + "textMode": "auto", + "wideLayout": true + }, + "pluginVersion": "11.0.0", + "targets": [ + { + "editorMode": "code", + "expr": "vllm:spec_decode_draft_tokens_per_second:rate1m", + "legendFormat": "Draft", + "range": true, + "refId": "A" + } + ], + "title": "Draft tokens / second", + "type": "stat" + }, + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "fieldConfig": { + "defaults": { + "decimals": 1, + "min": 0, + "unit": "suffix: tokens/s" + }, + "overrides": [] + }, + "gridPos": { + "h": 10, + "w": 24, + "x": 0, + "y": 6 + }, + "id": 4, + "options": { + "legend": { + "calcs": [ + "mean", + "lastNotNull" + ], + "displayMode": "table", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "pluginVersion": "11.0.0", + "targets": [ + { + "editorMode": "code", + "expr": "vllm:spec_decode_draft_tokens_per_second:rate1m", + "legendFormat": "Draft tokens", + "range": true, + "refId": "A" + }, + { + "editorMode": "code", + "expr": "vllm:spec_decode_accepted_tokens_per_second:rate1m", + "legendFormat": "Accepted tokens", + "range": true, + "refId": "B" + } + ], + "title": "Draft vs accepted tokens / second (1-minute average)", + "type": "timeseries" + }, + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "fieldConfig": { + "defaults": { + "decimals": 1, + "max": 1, + "min": 0, + "unit": "percentunit" + }, + "overrides": [] + }, + "gridPos": { + "h": 10, + "w": 24, + "x": 0, + "y": 16 + }, + "id": 5, + "options": { + "barRadius": 0, + "barWidth": 0.8, + "fullHighlight": false, + "groupWidth": 0.7, + "legend": { + "calcs": [], + "displayMode": "list", + "placement": "bottom", + "showLegend": false + }, + "orientation": "auto", + "showValue": "auto", + "stacking": "none", + "tooltip": { + "mode": "single", + "sort": "desc" + }, + "xTickLabel": "Position {{text}}" + }, + "pluginVersion": "11.0.0", + "targets": [ + { + "editorMode": "code", + "expr": "vllm:spec_decode_acceptance_rate_by_pos:rate1m", + "legendFormat": "Position {{position}}", + "range": true, + "refId": "A" + } + ], + "title": "Acceptance rate by draft position (1-minute average)", + "type": "barchart" + } + ], + "refresh": "5s", + "schemaVersion": 39, + "tags": [ + "vllm", + "prometheus", + "spec-decode" + ], + "templating": { + "list": [] + }, + "time": { + "from": "now-15m", + "to": "now" + }, + "timepicker": {}, + "timezone": "browser", + "title": "vLLM Speculative Decoding", + "uid": "vllm-spec-decode", + "version": 1, + "weekStart": "" +} diff --git a/prometheus-rules.yaml b/prometheus-rules.yaml index 788996a1..0864d65e 100644 --- a/prometheus-rules.yaml +++ b/prometheus-rules.yaml @@ -13,3 +13,51 @@ groups: sum(rate(vllm:generation_tokens_total{job="vllm"}[1m])) + sum(rate(vllm:prompt_tokens_total{job="vllm"}[1m])) + + - name: vllm-spec-decode + interval: 5s + rules: + - record: vllm:spec_decode_acceptance_rate:rate1m + expr: >- + ( + sum(rate(vllm:spec_decode_num_accepted_tokens_total{job="vllm"}[1m])) + / + sum(rate(vllm:spec_decode_num_draft_tokens_total{job="vllm"}[1m])) + ) + and on() + ( + sum(rate(vllm:spec_decode_num_draft_tokens_total{job="vllm"}[1m])) > 0 + ) + + - record: vllm:spec_decode_mean_accepted_length:rate1m + expr: >- + ( + 1 + + sum(rate(vllm:spec_decode_num_accepted_tokens_total{job="vllm"}[1m])) + / + sum(rate(vllm:spec_decode_num_drafts_total{job="vllm"}[1m])) + ) + and on() + ( + sum(rate(vllm:spec_decode_num_drafts_total{job="vllm"}[1m])) > 0 + ) + + - record: vllm:spec_decode_draft_tokens_per_second:rate1m + expr: sum(rate(vllm:spec_decode_num_draft_tokens_total{job="vllm"}[1m])) + + - record: vllm:spec_decode_accepted_tokens_per_second:rate1m + expr: sum(rate(vllm:spec_decode_num_accepted_tokens_total{job="vllm"}[1m])) + + - record: vllm:spec_decode_acceptance_rate_by_pos:rate1m + expr: >- + ( + sum by (position) ( + rate(vllm:spec_decode_num_accepted_tokens_per_pos_total{job="vllm"}[1m]) + ) + / + scalar(sum(rate(vllm:spec_decode_num_drafts_total{job="vllm"}[1m]))) + ) + and on() + ( + sum(rate(vllm:spec_decode_num_drafts_total{job="vllm"}[1m])) > 0 + ) From 4b62d712a6aaa1d13b8359d6a43be093f1d75162 Mon Sep 17 00:00:00 2001 From: Kaushalya Date: Mon, 17 Aug 2026 11:06:13 +0900 Subject: [PATCH 2/5] Enable MTP speculative decoding in qwen3.8-27b-nvfp4 recipe --- recipes/qwen3.8-27b-nvfp4.yaml | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/recipes/qwen3.8-27b-nvfp4.yaml b/recipes/qwen3.8-27b-nvfp4.yaml index 2bf409df..c6e7a155 100644 --- a/recipes/qwen3.8-27b-nvfp4.yaml +++ b/recipes/qwen3.8-27b-nvfp4.yaml @@ -1,6 +1,8 @@ # Recipe: Qwen3.8-27B-NVFP4 # Unsloth's mixed-precision NVFP4 quant of Qwen/Qwen3.8-27B. # A single DGX Spark has enough unified memory for the native 262K context. +# Native MTP requires vLLM with the Qwen GDN speculative-decoding fixes from +# vllm-project/vllm#51812 and #51674 (merged 2026-08-14 or later). recipe_version: "1" name: Qwen3.8-27B-NVFP4 @@ -24,6 +26,7 @@ defaults: max_model_len: 262144 max_num_seqs: 4 max_num_batched_tokens: 8192 + num_speculative_tokens: 3 env: {} @@ -41,6 +44,7 @@ command: | --load-format fastsafetensors \ --enable-chunked-prefill \ --enable-prefix-caching \ + --speculative-config '{{"method":"mtp","num_speculative_tokens":{num_speculative_tokens}}}' \ --enable-auto-tool-choice \ --tool-call-parser qwen3_xml \ --reasoning-parser qwen3 From cd5814ccdd1fa62dfe2425aebc20eea0733b20d6 Mon Sep 17 00:00:00 2001 From: Kaushalya Date: Mon, 17 Aug 2026 12:01:07 +0900 Subject: [PATCH 3/5] Query spec-decode stat panels as instant values --- docs/METRICS.md | 6 ++++-- grafana/dashboards/vllm-spec-decode.json | 16 ++++++++-------- 2 files changed, 12 insertions(+), 10 deletions(-) diff --git a/docs/METRICS.md b/docs/METRICS.md index 13e7535b..1d20049d 100644 --- a/docs/METRICS.md +++ b/docs/METRICS.md @@ -108,8 +108,10 @@ recording rules from the `vllm-spec-decode` group in `prometheus-rules.yaml`: - **Acceptance rate by draft position** — per-position quality of the MTP cascade (`vllm:spec_decode_acceptance_rate_by_pos:rate1m`) -The rules evaluate to no value while vLLM is idle, so the panels show **No -value** until requests are being generated; this avoids a misleading 100% +The stat panels query the recorded metrics as instant values. The rules +evaluate to no value once traffic stops, so the stats show **No value** +shortly after the last request, when the last recorded sample leaves +Prometheus' five-minute lookback window; this avoids a misleading 100% acceptance rate at idle. If you add or replace dashboard files on disk, restart the Grafana container diff --git a/grafana/dashboards/vllm-spec-decode.json b/grafana/dashboards/vllm-spec-decode.json index b50734c4..0ad48d11 100644 --- a/grafana/dashboards/vllm-spec-decode.json +++ b/grafana/dashboards/vllm-spec-decode.json @@ -32,7 +32,7 @@ "id": 1, "options": { "colorMode": "value", - "graphMode": "area", + "graphMode": "none", "justifyMode": "auto", "orientation": "auto", "reduceOptions": { @@ -52,7 +52,7 @@ "editorMode": "code", "expr": "vllm:spec_decode_acceptance_rate:rate1m", "legendFormat": "Accepted", - "range": true, + "range": false, "refId": "A" } ], @@ -81,7 +81,7 @@ "id": 2, "options": { "colorMode": "value", - "graphMode": "area", + "graphMode": "none", "justifyMode": "auto", "orientation": "auto", "reduceOptions": { @@ -101,7 +101,7 @@ "editorMode": "code", "expr": "vllm:spec_decode_mean_accepted_length:rate1m", "legendFormat": "Tokens / step", - "range": true, + "range": false, "refId": "A" } ], @@ -130,7 +130,7 @@ "id": 3, "options": { "colorMode": "value", - "graphMode": "area", + "graphMode": "none", "justifyMode": "auto", "orientation": "auto", "reduceOptions": { @@ -150,7 +150,7 @@ "editorMode": "code", "expr": "vllm:spec_decode_draft_tokens_per_second:rate1m", "legendFormat": "Draft", - "range": true, + "range": false, "refId": "A" } ], @@ -259,7 +259,7 @@ "editorMode": "code", "expr": "vllm:spec_decode_acceptance_rate_by_pos:rate1m", "legendFormat": "Position {{position}}", - "range": true, + "range": false, "refId": "A" } ], @@ -285,6 +285,6 @@ "timezone": "browser", "title": "vLLM Speculative Decoding", "uid": "vllm-spec-decode", - "version": 1, + "version": 2, "weekStart": "" } From 1238aaa0b1724579446cf1542a93dba8917db3a3 Mon Sep 17 00:00:00 2001 From: Kaushalya Date: Mon, 17 Aug 2026 12:30:40 +0900 Subject: [PATCH 4/5] Render per-position acceptance as a Bar gauge --- docs/METRICS.md | 5 +++-- grafana/dashboards/vllm-spec-decode.json | 20 +++++++------------- 2 files changed, 10 insertions(+), 15 deletions(-) diff --git a/docs/METRICS.md b/docs/METRICS.md index 1d20049d..f199d357 100644 --- a/docs/METRICS.md +++ b/docs/METRICS.md @@ -194,8 +194,9 @@ and on() ) ``` -Use a Bar chart or Time series visualization and set the legend to -`Position {{position}}`. +The provisioned dashboard renders this metric as a Bar gauge with one bar +per draft position. If you build your own panel, use a Bar gauge or Time +series visualization and set the legend to `Position {{position}}`. ## Generate test traffic diff --git a/grafana/dashboards/vllm-spec-decode.json b/grafana/dashboards/vllm-spec-decode.json index 0ad48d11..5ac99081 100644 --- a/grafana/dashboards/vllm-spec-decode.json +++ b/grafana/dashboards/vllm-spec-decode.json @@ -234,24 +234,18 @@ }, "id": 5, "options": { - "barRadius": 0, - "barWidth": 0.8, - "fullHighlight": false, - "groupWidth": 0.7, + "displayMode": "gradient", "legend": { "calcs": [], "displayMode": "list", "placement": "bottom", "showLegend": false }, + "namePlacement": "auto", "orientation": "auto", - "showValue": "auto", - "stacking": "none", - "tooltip": { - "mode": "single", - "sort": "desc" - }, - "xTickLabel": "Position {{text}}" + "showUnfilled": true, + "sizing": "auto", + "valueMode": "color" }, "pluginVersion": "11.0.0", "targets": [ @@ -264,7 +258,7 @@ } ], "title": "Acceptance rate by draft position (1-minute average)", - "type": "barchart" + "type": "bargauge" } ], "refresh": "5s", @@ -285,6 +279,6 @@ "timezone": "browser", "title": "vLLM Speculative Decoding", "uid": "vllm-spec-decode", - "version": 2, + "version": 3, "weekStart": "" } From b66a5535e344d5ec53b92036deec205d79d3539a Mon Sep 17 00:00:00 2001 From: Kaushalya Date: Mon, 17 Aug 2026 14:52:46 +0900 Subject: [PATCH 5/5] Document idle behavior of guarded vs unguarded spec-decode rules --- docs/METRICS.md | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/docs/METRICS.md b/docs/METRICS.md index f199d357..47202dd0 100644 --- a/docs/METRICS.md +++ b/docs/METRICS.md @@ -108,11 +108,14 @@ recording rules from the `vllm-spec-decode` group in `prometheus-rules.yaml`: - **Acceptance rate by draft position** — per-position quality of the MTP cascade (`vllm:spec_decode_acceptance_rate_by_pos:rate1m`) -The stat panels query the recorded metrics as instant values. The rules -evaluate to no value once traffic stops, so the stats show **No value** -shortly after the last request, when the last recorded sample leaves -Prometheus' five-minute lookback window; this avoids a misleading 100% -acceptance rate at idle. +The stat panels query the recorded metrics as instant values. The ratio +rules (acceptance rate, mean accepted length, per-position) evaluate to no +value once traffic stops, so those panels show **No value** shortly after +the last request, when the last recorded sample leaves Prometheus' +five-minute lookback window; this avoids a misleading 100% acceptance rate +at idle. The token-rate rules have no such guard and keep recording zero +while vLLM is running, so the draft and accepted token panels show 0 at +idle. If you add or replace dashboard files on disk, restart the Grafana container for the file provider to reload them: