Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions tools/server/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -1074,6 +1074,10 @@ In *router mode* the query param `?model={model_id}` has to be set. This endpoin
| `llamacpp:n_tokens_max` | Counter | High watermark of the context size observed. |
| `llamacpp:n_decode_total` | Counter | Total Number of llama_decode() calls. |
| `llamacpp:n_busy_slots_per_decode` | Gauge | Average number of busy slots per llama_decode() call. |
| `llamacpp:spec_decode_num_draft_tokens_total` | Counter | Total draft tokens generated (0 when spec-decode is off). |
| `llamacpp:spec_decode_num_accepted_tokens_total` | Counter | Total draft tokens accepted by the target model (0 when spec-decode is off). |
| `llamacpp:spec_decode_num_drafts_total` | Counter | Total speculative decoding verification steps (0 when spec-decode is off). |
| `llamacpp:spec_decode_num_accepted_tokens_per_pos_total` | Counter | Accepted tokens per draft position (labeled `position="N"`; absent when spec-decode is off or before the first completed speculative request). |

### POST `/slots/{id_slot}?action=save`: Save the prompt cache of the specified slot to a file.

Expand Down
44 changes: 44 additions & 0 deletions tools/server/server-context.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -845,6 +845,11 @@ struct server_metrics {
uint64_t n_decode_total = 0;
uint64_t n_busy_slots_total = 0;

uint64_t n_draft_tokens_total = 0;
uint64_t n_draft_accepted_total = 0;
uint64_t n_draft_verif_steps_total = 0;
std::vector<uint64_t> n_accepted_per_pos_total;

void init() {
t_start = ggml_time_us();
}
Expand All @@ -863,6 +868,17 @@ struct server_metrics {
n_tokens_predicted += slot.n_decoded;
t_tokens_generation += slot.t_token_generation;
t_tokens_generation_total += slot.t_token_generation;

n_draft_tokens_total += slot.n_draft_total;
n_draft_accepted_total += slot.n_draft_accepted;
n_draft_verif_steps_total += slot.n_draft_verif_steps;

if (n_accepted_per_pos_total.size() < slot.n_accepted_per_pos.size()) {
n_accepted_per_pos_total.resize(slot.n_accepted_per_pos.size(), 0);
}
for (size_t i = 0; i < slot.n_accepted_per_pos.size(); i++) {
n_accepted_per_pos_total[i] += slot.n_accepted_per_pos[i];
}
}

void on_decoded(const std::vector<server_slot> & slots) {
Expand Down Expand Up @@ -2552,6 +2568,11 @@ struct server_context_impl {
res->n_decode_total = metrics.n_decode_total;
res->n_busy_slots_total = metrics.n_busy_slots_total;

res->n_draft_tokens_total = metrics.n_draft_tokens_total;
res->n_draft_accepted_total = metrics.n_draft_accepted_total;
res->n_draft_verif_steps_total = metrics.n_draft_verif_steps_total;
res->n_accepted_per_pos_total = metrics.n_accepted_per_pos_total;

if (task.metrics_reset_bucket) {
metrics.reset_bucket();
}
Expand Down Expand Up @@ -4441,6 +4462,18 @@ void server_routes::init_routes() {
{"name", "n_tokens_max"},
{"help", "Largest observed n_tokens."},
{"value", res_task->n_tokens_max}
}, {
{"name", "spec_decode_num_draft_tokens_total"},
{"help", "Total draft tokens generated"},
{"value", res_task->n_draft_tokens_total}
}, {
{"name", "spec_decode_num_accepted_tokens_total"},
{"help", "Total draft tokens accepted by the target model"},
{"value", res_task->n_draft_accepted_total}
}, {
{"name", "spec_decode_num_drafts_total"},
{"help", "Total speculative decoding verification steps"},
{"value", res_task->n_draft_verif_steps_total}
}}},
{"gauge", {{
{"name", "prompt_tokens_seconds"},
Expand Down Expand Up @@ -4482,6 +4515,17 @@ void server_routes::init_routes() {
}
}

// labeled counter: one time series per draft position
if (!res_task->n_accepted_per_pos_total.empty()) {
prometheus << "# HELP llamacpp:spec_decode_num_accepted_tokens_per_pos_total"
" Accepted tokens per draft position\n"
<< "# TYPE llamacpp:spec_decode_num_accepted_tokens_per_pos_total counter\n";
for (size_t i = 0; i < res_task->n_accepted_per_pos_total.size(); i++) {
prometheus << "llamacpp:spec_decode_num_accepted_tokens_per_pos_total{position=\""
<< i << "\"} " << res_task->n_accepted_per_pos_total[i] << "\n";
}
}

res->headers["Process-Start-Time-Unix"] = std::to_string(res_task->t_start);
res->content_type = "text/plain; version=0.0.4";
res->status = 200;
Expand Down
5 changes: 5 additions & 0 deletions tools/server/server-task.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -1560,6 +1560,11 @@ json server_task_result_metrics::to_json() {
{ "n_decode_total", n_decode_total },
{ "n_busy_slots_total", n_busy_slots_total },

{ "n_draft_tokens_total", n_draft_tokens_total },
{ "n_draft_accepted_total", n_draft_accepted_total },
{ "n_draft_verif_steps_total", n_draft_verif_steps_total },
{ "n_accepted_per_pos_total", n_accepted_per_pos_total },

{ "slots", slots_data },
};
}
Expand Down
5 changes: 5 additions & 0 deletions tools/server/server-task.h
Original file line number Diff line number Diff line change
Expand Up @@ -532,6 +532,11 @@ struct server_task_result_metrics : server_task_result {
uint64_t n_decode_total = 0;
uint64_t n_busy_slots_total = 0;

uint64_t n_draft_tokens_total = 0;
uint64_t n_draft_accepted_total = 0;
uint64_t n_draft_verif_steps_total = 0;
std::vector<uint64_t> n_accepted_per_pos_total;

// while we can also use std::vector<server_slot> this requires copying the slot object which can be quite messy
// therefore, we use json to temporarily store the slot.to_json() result
json slots_data = json::array();
Expand Down
Loading