Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions crates/onnx-runtime-ep-cpu/src/kernels/matmul_nbits.rs
Original file line number Diff line number Diff line change
Expand Up @@ -392,13 +392,15 @@ enum PrefillFanOut {
///
/// Park latency is what makes the wide path safe above the threshold: 226 us of
/// worst-case wake-up is 0.25% of a 90 ms fan-out, and 45% of a 0.5 ms one.
#[cfg(target_arch = "x86_64")]
const WIDE_PREFILL_MACS: usize = 1 << 29;

/// Picks the prefill fan-out executor for `macs` of work, given the task
/// runtime's `lanes` and global Rayon's `wide` width.
///
/// Split out as a pure function so the policy is testable without a machine
/// that has SMT, and so the threshold has one place to be wrong.
#[cfg(target_arch = "x86_64")]
fn prefill_fan_out(macs: usize, lanes: usize, wide: usize) -> PrefillFanOut {
// Nothing to win from the wide path when it is not actually wider; prefer
// the runtime's cheaper dispatch.
Expand Down Expand Up @@ -538,6 +540,7 @@ const MIN_PREFILL_TASK_MACS: usize = 1 << 19;
///
/// Returns a *floor* the task runtime applies to its own partition; the runtime
/// still uses a larger grain when there are more columns than workers.
#[cfg(target_arch = "x86_64")]
fn prefill_column_grain(m: usize, k: usize, n: usize) -> usize {
let macs_per_column = m.saturating_mul(k);
if macs_per_column == 0 {
Expand Down Expand Up @@ -17298,6 +17301,7 @@ mod tests {
/// A prefill small enough that the task runtime's ~5 us dispatch dominates
/// stays on the task runtime, whatever the widths look like.
#[test]
#[cfg(target_arch = "x86_64")]
fn small_prefill_work_stays_on_the_task_runtime() {
assert_eq!(
prefill_fan_out(WIDE_PREFILL_MACS - 1, 16, 32),
Expand All @@ -17310,6 +17314,7 @@ mod tests {
/// on work long enough for a 226 us wake-up to be noise, so take the wide
/// path.
#[test]
#[cfg(target_arch = "x86_64")]
fn large_prefill_work_takes_the_wide_fan_out() {
assert_eq!(
prefill_fan_out(WIDE_PREFILL_MACS, 16, 32),
Expand All @@ -17325,6 +17330,7 @@ mod tests {
/// have. When it has them -- no SMT, an explicit task-thread budget, a
/// narrow cpuset -- the cheaper dispatch wins unconditionally.
#[test]
#[cfg(target_arch = "x86_64")]
fn the_wide_fan_out_is_not_taken_when_it_is_not_wider() {
for wide in 1..=16 {
assert_eq!(
Expand Down Expand Up @@ -17367,6 +17373,7 @@ mod tests {
/// The native fan-out's grain is a floor in *output columns*, so it must
/// never exceed the column count nor drop below one.
#[test]
#[cfg(target_arch = "x86_64")]
fn prefill_column_grain_stays_within_the_column_count() {
for &(m, k, n) in &[
(1usize, 1usize, 1usize),
Expand All @@ -17383,6 +17390,7 @@ mod tests {
/// A column that already carries enough arithmetic is handed out one per
/// task; thinner columns get batched until they clear the floor.
#[test]
#[cfg(target_arch = "x86_64")]
fn prefill_column_grain_batches_only_undersized_columns() {
// 8 x 1024 = 8 Ki MACs a column, so batch 64 of them to clear 512 Ki.
assert_eq!(prefill_column_grain(8, 1024, 3072), 64);
Expand All @@ -17394,6 +17402,7 @@ mod tests {

/// A degenerate shape must not divide by zero or ask for a zero grain.
#[test]
#[cfg(target_arch = "x86_64")]
fn prefill_column_grain_survives_a_zero_sized_problem() {
assert_eq!(prefill_column_grain(0, 1024, 8), 8);
assert_eq!(prefill_column_grain(8, 0, 8), 8);
Expand Down
23 changes: 22 additions & 1 deletion crates/onnx-runtime-ep-cuda/src/weight_paging.rs
Original file line number Diff line number Diff line change
Expand Up @@ -85,7 +85,21 @@ static GLOBAL_VRAM_FREE_NS: AtomicU64 = AtomicU64::new(0);
// consumers of the VA to finish — but under VMM over-subscription that in-flight
// work is itself PCIe-fault-slowed, so folding it into `GLOBAL_VRAM_FREE_NS`
// mis-attributes paging-stalled compute/copy as "free" time (#1295). Timed
// separately so `vram_free_ns` measures only unmap/release.
// separately so the stream drain is excluded from `vram_free_ns`.
//
// CAUTION (2026-08-19 reconciliation, docs/benchmarks/
// 2026-08-19-vram-free-attribution-reconciliation.md): even with the drain
// split out, `GLOBAL_VRAM_FREE_NS` is NOT "driver freeing time". It wraps the
// whole free code path -- the `cuMemUnmap`/`cuMemRelease` driver calls AND the
// per-granule Rust bookkeeping in `decommit_allocation_range`/`deallocate_span`
// -- and the bookkeeping dominates: on an RTX 4060 weight-lending run the driver
// `cuMemUnmap` was a stable ~16.9 ms/call (~101 ms total over 6 releases) while
// `vram_free_ms` swung 82 ms <-> 2450 ms (~30x) on byte-identical deterministic
// work. A metric that varies 30x while the work is constant is measuring
// variable non-driver time (bookkeeping and/or lock/blocking), not freeing.
// To isolate the driver cost, time the `cuMemUnmap` sites directly (see the
// reconciliation doc's split instrumentation); do not read `vram_free_ms` as a
// driver-freeing figure.
static GLOBAL_VRAM_FREE_SYNC_NS: AtomicU64 = AtomicU64::new(0);
// Process-lifetime high-water gauge. Resetting activity counters must not write
// it: a concurrent page-in could otherwise be overwritten with a stale value.
Expand Down Expand Up @@ -177,6 +191,13 @@ pub struct GlobalOffloadStats {
pub materialize_fallback_calls: u64,
pub htod_bytes: u64,
pub vram_alloc_ns: u64,
/// Weight-page free code path: the `cuMemUnmap`/`cuMemRelease` driver calls
/// plus the per-granule Rust bookkeeping in `decommit_allocation_range`/
/// `deallocate_span`. NOT a driver-freeing figure -- the bookkeeping
/// dominates and this counter swings ~30x on byte-identical work while the
/// driver `cuMemUnmap` stays ~17 ms/call (see `GLOBAL_VRAM_FREE_NS` and the
/// 2026-08-19 reconciliation doc). Excludes the pre-free stream drain, which
/// is timed into [`Self::vram_free_sync_ns`].
pub vram_free_ns: u64,
/// Pre-free stream drain time (`cuStreamSynchronize`) taken on the Drop-path
/// eviction before unmapping. Split out of [`Self::vram_free_ns`] so the
Expand Down
Loading