Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
67 commits
Select commit Hold shift + click to select a range
f17f446
Replace Skippy verify span with verify windows
i386 Jul 1, 2026
09c9e7b
Add verify window reply metadata
i386 Jul 1, 2026
b0142ad
Merge remote-tracking branch 'origin/main' into experiment/skippy-pip…
i386 Jul 10, 2026
259661e
Pipeline direct-return n-gram verify windows
i386 Jul 10, 2026
2c6256e
Pipeline MTP-anchored n-gram verify windows
i386 Jul 14, 2026
ce9ddec
Merge remote-tracking branch 'origin/main' into experiment/skippy-pip…
i386 Jul 14, 2026
0f7ff0a
Merge remote-tracking branch 'origin/main' into experiment/skippy-pip…
i386 Jul 14, 2026
2f87be4
Support static release builds without features
i386 Jul 14, 2026
f0caba3
Merge remote-tracking branch 'origin/main' into experiment/skippy-pip…
i386 Jul 14, 2026
4e912a0
Merge branch 'main' into experiment/skippy-pipelined-decode
i386 Jul 14, 2026
1def9b7
Fix split MTP activation-frame serving
i386 Jul 14, 2026
8706428
Merge remote-tracking branch 'origin/experiment/skippy-pipelined-deco…
i386 Jul 14, 2026
e2d7fd4
Merge remote-tracking branch 'origin/main' into experiment/skippy-pip…
i386 Jul 15, 2026
e3f2054
Replace native MTP batched verifier with verify windows
i386 Jul 15, 2026
2926aee
Restore native MTP verify window batching
i386 Jul 15, 2026
6c37e6f
Replace native MTP anchor extension with composite proposals
i386 Jul 15, 2026
ab81efe
Keep composite decode branch on development version
i386 Jul 15, 2026
b9b4139
Expose decode timings for all generation modes
i386 Jul 15, 2026
f475971
Retry transient staged lane readiness
i386 Jul 16, 2026
051220a
Bound persistent lane readiness handshake
i386 Jul 16, 2026
a8568b5
Keep pure N-gram decode free of MTP drafts
i386 Jul 16, 2026
baa6c86
Report composite proposal totals in decode timings
i386 Jul 16, 2026
7fe4b95
Gate composite decode pipeline by candidate depth
i386 Jul 16, 2026
e98a42a
Account direct GGUF MTP weights in split planning
i386 Jul 16, 2026
ce1bb37
Avoid MTP cooldown after N-gram tail rejection
i386 Jul 16, 2026
39d570c
Improve hybrid MTP verification telemetry
i386 Jul 16, 2026
2a3f0ea
Pipeline native MTP verification replies
i386 Jul 16, 2026
1f597c2
Require useful N-gram tails for hybrid MTP
i386 Jul 16, 2026
4488411
Adapt N-gram MTP extensions to tail acceptance
i386 Jul 16, 2026
74b7f6e
Fix direct GGUF planning fallback
i386 Jul 16, 2026
96a9948
Gate N-gram tails on MTP prefix agreement
i386 Jul 16, 2026
4878187
Widen initial async verify windows
i386 Jul 16, 2026
9dd6129
Restore anchored N-gram MTP extensions
i386 Jul 16, 2026
0e28afd
Merge remote-tracking branch 'origin/main' into experiment/skippy-pip…
i386 Jul 16, 2026
e980348
Retain ready stages across transient refresh failures
i386 Jul 16, 2026
0a1f356
Document pipelined VerifyWindow decode
i386 Jul 16, 2026
e45be27
Use llama.cpp N-gram proposer for Skippy
i386 Jul 16, 2026
a45d253
Add cache-based N-gram proposer
i386 Jul 16, 2026
7d050f3
Add declarative speculative proposer package schema
i386 Jul 16, 2026
13f6a37
Productize Skippy speculative decode plans
i386 Jul 17, 2026
017ba4d
Productize Skippy speculative decode plans
i386 Jul 17, 2026
9229a4b
Support direct N-gram Skippy plans
i386 Jul 17, 2026
74de825
Validate speculative package strategy plans
i386 Jul 17, 2026
43cf106
Add coding agent loop benchmark corpus
i386 Jul 17, 2026
078edbc
Expose Skippy speculative benchmark counters
i386 Jul 17, 2026
30a2d1e
Validate cache N-gram proposer limits
i386 Jul 17, 2026
6b290a7
Merge remote-tracking branch 'origin/main' into experiment/skippy-pip…
i386 Jul 17, 2026
8f2f91c
Document speculative decode configuration
i386 Jul 17, 2026
3f5ed1c
Fix native MTP proposals and fused restore routing
i386 Jul 17, 2026
302cdc9
Honor configured N-gram extension width
i386 Jul 17, 2026
30ac739
Document speculative runtime overrides
i386 Jul 17, 2026
bfb174e
Refresh speculative config schema contracts
i386 Jul 17, 2026
ce0330e
Keep N-gram tail rejects from penalizing MTP
i386 Jul 17, 2026
c823ed1
Merge remote-tracking branch 'origin/main' into experiment/skippy-pip…
i386 Jul 17, 2026
5670251
Preserve MTP state after serial tail rejects
i386 Jul 17, 2026
b6d8c59
Report adaptive verify width changes accurately
i386 Jul 17, 2026
e1f1f99
Fix short simple N-gram extension budgets
i386 Jul 17, 2026
146a860
Make VerifyWindow pipelining cost-aware
i386 Jul 17, 2026
2678813
Profile prospective VerifyWindow widths
i386 Jul 17, 2026
a3ff7ac
Merge origin/main into experiment/skippy-pipelined-decode
michaelneale Jul 18, 2026
83a5653
docs: WAN split performance model + measured latency/compute decompos…
michaelneale Jul 18, 2026
572e5b8
docs: plan for fast-fail on new requests routed to a dead split stage
michaelneale Jul 18, 2026
dd63873
Fast-fail lane reconnects so new requests don't hang on a dead split …
michaelneale Jul 18, 2026
3aadab1
docs: latency-aware placement — current behaviour and many-node gaps
michaelneale Jul 18, 2026
b650142
docs: measured speculative recovery cost over WAN (why ngram hurts a …
michaelneale Jul 19, 2026
dc2e98a
Discard dead pooled stage lanes before reuse (fast-fail improvement)
michaelneale Jul 19, 2026
0972bc7
Merge remote-tracking branch 'origin/main' into experiment/skippy-pip…
michaelneale Jul 19, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

5 changes: 3 additions & 2 deletions crates/mesh-llm-cli/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@ pub use mesh_llm_events::LogFormat;
pub use parser::{
AuthCommand, BinaryFlavor, Cli, Command, ConfigCommand, DiscoveryScope, DoctorCommand,
GpuCommand, MeshDiscoveryMode, MeshGuardrailCliMode, NormalizedRuntimeArgs, PluginCommand,
RuntimeSurface, SkillAgentArg, SkillCommand, TrustCommand, TrustPolicy,
legacy_runtime_surface_warning, normalize_runtime_surface_args, validate_discovery_mode_args,
RuntimeSurface, SkillAgentArg, SkillCommand, SpeculativeNgramProposerCli, TrustCommand,
TrustPolicy, legacy_runtime_surface_warning, normalize_runtime_surface_args,
validate_discovery_mode_args,
};
112 changes: 112 additions & 0 deletions crates/mesh-llm-cli/src/parser.rs
Original file line number Diff line number Diff line change
Expand Up @@ -395,6 +395,21 @@ pub enum MeshGuardrailCliMode {
Enforce,
}

#[derive(Clone, Copy, Debug, Eq, PartialEq, ValueEnum)]
pub enum SpeculativeNgramProposerCli {
Simple,
Cache,
}

impl SpeculativeNgramProposerCli {
pub fn as_str(self) -> &'static str {
match self {
Self::Simple => "simple",
Self::Cache => "cache",
}
}
}

impl MeshGuardrailCliMode {
pub const fn as_str(self) -> &'static str {
match self {
Expand Down Expand Up @@ -547,6 +562,70 @@ pub struct Cli {
#[arg(long, hide = true)]
pub no_draft: bool,

/// Override the package speculative decoding strategy for this invocation.
#[arg(long, hide = true)]
pub speculative_strategy: Option<String>,

/// Override the N-gram proposer kind for this invocation.
#[arg(long, value_enum, hide = true)]
pub speculative_ngram_proposer: Option<SpeculativeNgramProposerCli>,

/// Minimum matching N-gram length for a direct N-gram proposer.
#[arg(long, hide = true)]
pub speculative_ngram_min: Option<u32>,

/// Maximum matching N-gram length for a direct N-gram proposer.
#[arg(long, hide = true)]
pub speculative_ngram_max: Option<u32>,

/// Cap N-gram tokens proposed in one verify window.
#[arg(long, hide = true)]
pub speculative_ngram_max_proposal_tokens: Option<u32>,

/// Initial N-gram extension length for a composite MTP strategy.
#[arg(long, hide = true)]
pub speculative_extension_initial_tokens: Option<u32>,

/// Maximum N-gram extension length for a composite MTP strategy.
#[arg(long, hide = true)]
pub speculative_extension_max_tokens: Option<u32>,

/// Consecutive weak extensions before the composite strategy backs off.
#[arg(long, hide = true)]
pub speculative_extension_tail_backoff_proposals: Option<u32>,

/// Native MTP rejection cooldown in generated tokens.
#[arg(long, hide = true)]
pub speculative_native_mtp_reject_cooldown_tokens: Option<u32>,

/// Suppress native MTP drafts while its rejection cooldown is active.
#[arg(long, hide = true)]
pub speculative_native_mtp_suppress_cooldown_drafts: bool,

/// Keep native MTP drafts during its rejection cooldown.
#[arg(
long,
hide = true,
conflicts_with = "speculative_native_mtp_suppress_cooldown_drafts"
)]
pub speculative_native_mtp_allow_cooldown_drafts: bool,

/// Maximum native MTP drafts suppressed by a cooldown.
#[arg(long, hide = true)]
pub speculative_native_mtp_suppress_cooldown_draft_limit: Option<u32>,

/// Minimum tokens to include in a pipelined verify window.
#[arg(long, hide = true)]
pub speculative_verify_window_min_tokens: Option<u32>,

/// Maximum tokens to include in a pipelined verify window.
#[arg(long, hide = true)]
pub speculative_verify_window_max_tokens: Option<u32>,

/// Number of in-flight pipelined verify windows.
#[arg(long, hide = true)]
pub speculative_verify_window_pipeline_depth: Option<u32>,

/// Force tensor split even if the model fits on one node.
#[arg(long, hide = true)]
pub split: bool,
Expand Down Expand Up @@ -1508,6 +1587,39 @@ mod tests {
);
}

#[test]
fn serve_parses_speculative_decode_overrides() {
let normalized = normalize_runtime_surface_args([
"mesh-llm",
"serve",
"--speculative-strategy",
"mtp-cache",
"--speculative-ngram-proposer",
"cache",
"--speculative-ngram-min",
"2",
"--speculative-ngram-max",
"6",
"--speculative-extension-max-tokens",
"8",
"--speculative-native-mtp-allow-cooldown-drafts",
"--speculative-verify-window-pipeline-depth",
"3",
]);
let cli = Cli::try_parse_from(normalized.normalized).expect("clap parse");

assert_eq!(cli.speculative_strategy.as_deref(), Some("mtp-cache"));
assert_eq!(
cli.speculative_ngram_proposer,
Some(SpeculativeNgramProposerCli::Cache)
);
assert_eq!(cli.speculative_ngram_min, Some(2));
assert_eq!(cli.speculative_ngram_max, Some(6));
assert_eq!(cli.speculative_extension_max_tokens, Some(8));
assert!(cli.speculative_native_mtp_allow_cooldown_drafts);
assert_eq!(cli.speculative_verify_window_pipeline_depth, Some(3));
}

#[test]
fn legacy_runtime_surface_warning_for_top_level_serve_flags() {
let normalized =
Expand Down
30 changes: 27 additions & 3 deletions crates/mesh-llm-config/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -38,8 +38,8 @@ pub use validate::{
mod tests {
use super::{
ConfigStore, GpuAssignment, LocalServingNodeConfig, MeshConfig, ModelRuntimeKind,
built_in_config_schema, canonicalize_built_in_config_identifier, parse_config_toml,
validate_config,
SpeculativeConfig, built_in_config_schema, canonicalize_built_in_config_identifier,
parse_config_toml, validate_config,
};
use std::collections::{BTreeMap, BTreeSet};
use std::fs;
Expand All @@ -55,6 +55,31 @@ mod tests {
assert!(config.models.is_empty());
}

#[test]
fn speculative_config_precedence_keeps_lower_layer_fields() {
let defaults = SpeculativeConfig {
strategy: Some("mtp-cache".to_string()),
verify_window_pipeline_depth: Some(2),
..Default::default()
};
let model = SpeculativeConfig {
ngram_max_proposal_tokens: Some(6),
..Default::default()
};
let overrides = SpeculativeConfig {
strategy: Some("mtp".to_string()),
verify_window_pipeline_depth: Some(3),
..Default::default()
};

let resolved =
SpeculativeConfig::with_precedence(Some(&overrides), Some(&model), Some(&defaults));

assert_eq!(resolved.strategy.as_deref(), Some("mtp"));
assert_eq!(resolved.ngram_max_proposal_tokens, Some(6));
assert_eq!(resolved.verify_window_pipeline_depth, Some(3));
}

#[test]
fn plugin_startup_config_round_trips_from_toml() {
let config: MeshConfig = toml::from_str(
Expand Down Expand Up @@ -731,7 +756,6 @@ gpu_id = "pci:0000:65:00.0"
"models",
"plugins",
"settings",
"strategy",
];

let mut total = 0usize;
Expand Down
131 changes: 130 additions & 1 deletion crates/mesh-llm-config/src/model.rs
Original file line number Diff line number Diff line change
Expand Up @@ -559,10 +559,80 @@ pub struct SpeculativeConfig {
pub draft_cache_type_v: Option<String>,
pub ngram_min: Option<u32>,
pub ngram_max: Option<u32>,
pub ngram_proposer: Option<String>,
pub ngram_max_proposal_tokens: Option<u32>,
pub extension_initial_tokens: Option<u32>,
pub extension_max_tokens: Option<u32>,
pub extension_tail_backoff_proposals: Option<u32>,
pub native_mtp_reject_cooldown_tokens: Option<u32>,
pub native_mtp_suppress_cooldown_drafts: Option<bool>,
pub native_mtp_suppress_cooldown_draft_limit: Option<u32>,
pub verify_window_min_tokens: Option<u32>,
pub verify_window_max_tokens: Option<u32>,
pub verify_window_pipeline_depth: Option<u32>,
pub spec_default: Option<BoolOrAuto>,
pub(crate) legacy_draft_model_path_used: bool,
}

impl SpeculativeConfig {
/// Resolves the three supported policy layers without discarding fields
/// that are not overridden by a more specific layer.
pub fn with_precedence(
overrides: Option<&Self>,
model: Option<&Self>,
defaults: Option<&Self>,
) -> Self {
macro_rules! pick {
($field:ident) => {
overrides
.and_then(|config| config.$field.clone())
.or_else(|| model.and_then(|config| config.$field.clone()))
.or_else(|| defaults.and_then(|config| config.$field.clone()))
};
}

Self {
strategy: pick!(strategy),
mode: pick!(mode),
draft_model: pick!(draft_model),
draft_hf_repo: pick!(draft_hf_repo),
draft_hf_file: pick!(draft_hf_file),
draft_selection_policy: pick!(draft_selection_policy),
pairing_fault: pick!(pairing_fault),
draft_max_tokens: pick!(draft_max_tokens),
draft_min_tokens: pick!(draft_min_tokens),
draft_acceptance_threshold: pick!(draft_acceptance_threshold),
draft_split_probability: pick!(draft_split_probability),
draft_gpu_layers: pick!(draft_gpu_layers),
draft_device: pick!(draft_device),
draft_threads: pick!(draft_threads),
draft_cache_type_k: pick!(draft_cache_type_k),
draft_cache_type_v: pick!(draft_cache_type_v),
ngram_min: pick!(ngram_min),
ngram_max: pick!(ngram_max),
ngram_proposer: pick!(ngram_proposer),
ngram_max_proposal_tokens: pick!(ngram_max_proposal_tokens),
extension_initial_tokens: pick!(extension_initial_tokens),
extension_max_tokens: pick!(extension_max_tokens),
extension_tail_backoff_proposals: pick!(extension_tail_backoff_proposals),
native_mtp_reject_cooldown_tokens: pick!(native_mtp_reject_cooldown_tokens),
native_mtp_suppress_cooldown_drafts: pick!(native_mtp_suppress_cooldown_drafts),
native_mtp_suppress_cooldown_draft_limit: pick!(
native_mtp_suppress_cooldown_draft_limit
),
verify_window_min_tokens: pick!(verify_window_min_tokens),
verify_window_max_tokens: pick!(verify_window_max_tokens),
verify_window_pipeline_depth: pick!(verify_window_pipeline_depth),
spec_default: pick!(spec_default),
legacy_draft_model_path_used: overrides
.filter(|config| config.draft_model.is_some())
.or_else(|| model.filter(|config| config.draft_model.is_some()))
.or_else(|| defaults.filter(|config| config.draft_model.is_some()))
.is_some_and(|config| config.legacy_draft_model_path_used),
}
}
}

/// Raw deserialization helper that accepts both `draft_model` and the legacy
/// `draft_model_path` key. The public `SpeculativeConfig` is constructed from
/// this after detecting which key was used.
Expand Down Expand Up @@ -608,6 +678,28 @@ struct SpeculativeConfigRaw {
#[serde(default)]
ngram_max: Option<u32>,
#[serde(default)]
ngram_proposer: Option<String>,
#[serde(default)]
ngram_max_proposal_tokens: Option<u32>,
#[serde(default)]
extension_initial_tokens: Option<u32>,
#[serde(default)]
extension_max_tokens: Option<u32>,
#[serde(default)]
extension_tail_backoff_proposals: Option<u32>,
#[serde(default)]
native_mtp_reject_cooldown_tokens: Option<u32>,
#[serde(default)]
native_mtp_suppress_cooldown_drafts: Option<bool>,
#[serde(default)]
native_mtp_suppress_cooldown_draft_limit: Option<u32>,
#[serde(default)]
verify_window_min_tokens: Option<u32>,
#[serde(default)]
verify_window_max_tokens: Option<u32>,
#[serde(default)]
verify_window_pipeline_depth: Option<u32>,
#[serde(default)]
spec_default: Option<BoolOrAuto>,
}

Expand Down Expand Up @@ -643,6 +735,17 @@ impl<'de> Deserialize<'de> for SpeculativeConfig {
draft_cache_type_v: raw.draft_cache_type_v,
ngram_min: raw.ngram_min,
ngram_max: raw.ngram_max,
ngram_proposer: raw.ngram_proposer,
ngram_max_proposal_tokens: raw.ngram_max_proposal_tokens,
extension_initial_tokens: raw.extension_initial_tokens,
extension_max_tokens: raw.extension_max_tokens,
extension_tail_backoff_proposals: raw.extension_tail_backoff_proposals,
native_mtp_reject_cooldown_tokens: raw.native_mtp_reject_cooldown_tokens,
native_mtp_suppress_cooldown_drafts: raw.native_mtp_suppress_cooldown_drafts,
native_mtp_suppress_cooldown_draft_limit: raw.native_mtp_suppress_cooldown_draft_limit,
verify_window_min_tokens: raw.verify_window_min_tokens,
verify_window_max_tokens: raw.verify_window_max_tokens,
verify_window_pipeline_depth: raw.verify_window_pipeline_depth,
spec_default: raw.spec_default,
legacy_draft_model_path_used: legacy_used,
})
Expand All @@ -656,7 +759,7 @@ impl Serialize for SpeculativeConfig {
{
use serde::ser::SerializeMap;

let mut map = serializer.serialize_map(Some(21))?;
let mut map = serializer.serialize_map(Some(32))?;
map.serialize_entry("strategy", &self.strategy)?;
map.serialize_entry("mode", &self.mode)?;
if self.legacy_draft_model_path_used {
Expand Down Expand Up @@ -684,6 +787,32 @@ impl Serialize for SpeculativeConfig {
map.serialize_entry("draft_cache_type_v", &self.draft_cache_type_v)?;
map.serialize_entry("ngram_min", &self.ngram_min)?;
map.serialize_entry("ngram_max", &self.ngram_max)?;
map.serialize_entry("ngram_proposer", &self.ngram_proposer)?;
map.serialize_entry("ngram_max_proposal_tokens", &self.ngram_max_proposal_tokens)?;
map.serialize_entry("extension_initial_tokens", &self.extension_initial_tokens)?;
map.serialize_entry("extension_max_tokens", &self.extension_max_tokens)?;
map.serialize_entry(
"extension_tail_backoff_proposals",
&self.extension_tail_backoff_proposals,
)?;
map.serialize_entry(
"native_mtp_reject_cooldown_tokens",
&self.native_mtp_reject_cooldown_tokens,
)?;
map.serialize_entry(
"native_mtp_suppress_cooldown_drafts",
&self.native_mtp_suppress_cooldown_drafts,
)?;
map.serialize_entry(
"native_mtp_suppress_cooldown_draft_limit",
&self.native_mtp_suppress_cooldown_draft_limit,
)?;
map.serialize_entry("verify_window_min_tokens", &self.verify_window_min_tokens)?;
map.serialize_entry("verify_window_max_tokens", &self.verify_window_max_tokens)?;
map.serialize_entry(
"verify_window_pipeline_depth",
&self.verify_window_pipeline_depth,
)?;
map.serialize_entry("spec_default", &self.spec_default)?;
map.end()
}
Expand Down
Loading
Loading