Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
24 changes: 23 additions & 1 deletion common/arg.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -1357,6 +1357,28 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
params.cache_ram_mib = value;
}
).set_env("LLAMA_ARG_CACHE_RAM").set_examples({LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_CLI}));
add_opt(common_arg(
{"--cache-disk"}, "PATH",
"base directory for the automatic SSD-backed prompt cache (default: disabled); "
"the server creates and removes an owner-only run directory below PATH",
[](common_params & params, const std::string & value) {
if (value.empty()) {
throw std::invalid_argument("cache disk path must not be empty");
}
params.cache_disk_path = value;
}
).set_env("LLAMA_ARG_CACHE_DISK").set_examples({LLAMA_EXAMPLE_SERVER}));
add_opt(common_arg(
{"--cache-disk-limit"}, "N",
string_format("maximum SSD-backed prompt-cache size in MiB when --cache-disk is set "
"(default: %d, 0 - disable)", params.cache_disk_limit_mib),
[](common_params & params, int value) {
if (value < 0) {
throw std::invalid_argument("cache disk limit must be non-negative");
}
params.cache_disk_limit_mib = value;
}
).set_env("LLAMA_ARG_CACHE_DISK_LIMIT").set_examples({LLAMA_EXAMPLE_SERVER}));
add_opt(common_arg(
{"-kvu", "--kv-unified"},
{"-no-kvu", "--no-kv-unified"},
Expand All @@ -1368,7 +1390,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
add_opt(common_arg(
{"--cache-idle-slots"},
{"--no-cache-idle-slots"},
"save idle slots to the prompt cache on new task, and clear them when using unified KV (default: enabled, requires cache-ram)",
"save idle slots to the prompt cache on new task, and clear them when using unified KV (default: enabled, requires cache RAM or disk)",
[](common_params & params, bool value) {
params.cache_idle_slots = value;
}
Expand Down
5 changes: 5 additions & 0 deletions common/common.h
Original file line number Diff line number Diff line change
Expand Up @@ -610,6 +610,7 @@ struct common_params {
int32_t n_ctx_checkpoints = 32; // max number of context checkpoints per slot
int32_t checkpoint_every_nt = 8192; // make a checkpoint every n tokens during prefill
int32_t cache_ram_mib = 8192; // -1 = no limit, 0 - disable, 1 = 1 MiB, etc.
int32_t cache_disk_limit_mib = 8192; // disk prompt-cache limit when cache_disk_path is set

std::string hostname = "127.0.0.1";
std::string public_path = ""; // NOLINT
Expand Down Expand Up @@ -647,6 +648,10 @@ struct common_params {
// enable built-in tools
std::vector<std::string> server_tools;

// Optional base directory for the automatic SSD-backed prompt cache. The
// server creates and owns a private per-process directory below this path.
std::string cache_disk_path;

// router server configs
std::string models_dir = ""; // directory containing models for the router server
std::string models_preset = ""; // directory containing model presets for the router server
Expand Down
307 changes: 296 additions & 11 deletions common/speculative.cpp

Large diffs are not rendered by default.

6 changes: 5 additions & 1 deletion common/speculative.h
Original file line number Diff line number Diff line change
Expand Up @@ -70,7 +70,11 @@ void common_speculative_accept(common_speculative * spec, llama_seq_id, uint16_t

// (optional) get/set internal state
bool common_speculative_get_state(common_speculative * spec, llama_seq_id seq_id, std::vector<uint8_t> & data);
void common_speculative_set_state(common_speculative * spec, llama_seq_id seq_id, const std::vector<uint8_t> & data);
bool common_speculative_set_state(common_speculative * spec, llama_seq_id seq_id, const std::vector<uint8_t> & data);
bool common_speculative_state_required(const common_speculative * spec);

// rebase per-sequence positions after the corresponding target/draft contexts shift
void common_speculative_shift_state(common_speculative * spec, llama_seq_id seq_id, llama_pos delta);

// print statistics about the speculative decoding
void common_speculative_print_stats(const common_speculative * spec);
Expand Down
4 changes: 3 additions & 1 deletion ggml/include/ggml.h
Original file line number Diff line number Diff line change
Expand Up @@ -436,7 +436,8 @@ extern "C" {
GGML_TYPE_Q3_0_ROCMFPX = 104, // ROCmFPx experimental 3-bit UE4M3-scale reference layout
GGML_TYPE_TURBO3_0 = 105, // TurboQuant 3-bit KV-cache (3.5 bpw)
GGML_TYPE_TURBO4_0 = 106, // TurboQuant 4-bit KV-cache (4.5 bpw)
GGML_TYPE_COUNT = 107,
GGML_TYPE_Q2_0_ROCMFPX = 107, // ROCmFPx experimental 2-bit S40 codebook + dual UE4M3 scales
GGML_TYPE_COUNT = 108,
};

// precision
Expand Down Expand Up @@ -490,6 +491,7 @@ extern "C" {
GGML_FTYPE_MOSTLY_Q6_0_ROCMFPX = 110, // ROCmFPx experimental 6-bit reference layout
GGML_FTYPE_MOSTLY_Q8_0_ROCMFPX = 111, // ROCmFPx experimental 8-bit reference layout
GGML_FTYPE_MOSTLY_Q3_0_ROCMFPX = 112, // ROCmFPx experimental 3-bit reference layout
GGML_FTYPE_MOSTLY_Q2_0_ROCMFPX = 113, // ROCmFPx experimental 2-bit S40 codebook layout
};

// available tensor operations:
Expand Down
Loading
Loading