From 29b058301975f1b460513e5c5e9a6e9bfd202c6a Mon Sep 17 00:00:00 2001 From: justinchuby <223556219+Copilot@users.noreply.github.com> Date: Wed, 19 Aug 2026 02:04:31 -0700 Subject: [PATCH 1/2] docs(cuda): correct vram_free_ns attribution comment (not driver-freeing time) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The `GLOBAL_VRAM_FREE_NS` / `vram_free_ns` counter previously documented itself as measuring "only unmap/release". The 2026-08-19 reconciliation (docs/benchmarks/2026-08-19-vram-free-attribution-reconciliation.md) showed that is misleading: with the stream drain already split into `vram_free_sync_ns`, the counter still wraps the whole free code path — the `cuMemUnmap`/`cuMemRelease` driver calls PLUS the per-granule Rust bookkeeping in `decommit_allocation_range`/`deallocate_span` — and the bookkeeping dominates. On an RTX 4060 weight-lending run the driver `cuMemUnmap` was a stable ~16.9 ms/call (~101 ms over 6 releases) while `vram_free_ms` swung 82 ms <-> 2450 ms (~30x) on byte-identical deterministic work. A metric that varies 30x while the work is constant is measuring variable non-driver time, not freeing. Documentation-only: updates the comment at the static definition and the `GlobalOffloadStats::vram_free_ns` field doc so the next reader does not mistake this counter for a driver-freeing figure. No behaviour, symbol, or output-schema change (the `vram_free_ms` CSV column is consumed by benchmark tooling and is intentionally left unrenamed). Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .../onnx-runtime-ep-cuda/src/weight_paging.rs | 23 ++++++++++++++++++- 1 file changed, 22 insertions(+), 1 deletion(-) diff --git a/crates/onnx-runtime-ep-cuda/src/weight_paging.rs b/crates/onnx-runtime-ep-cuda/src/weight_paging.rs index 11b6508832..669c7156ce 100644 --- a/crates/onnx-runtime-ep-cuda/src/weight_paging.rs +++ b/crates/onnx-runtime-ep-cuda/src/weight_paging.rs @@ -85,7 +85,21 @@ static GLOBAL_VRAM_FREE_NS: AtomicU64 = AtomicU64::new(0); // consumers of the VA to finish — but under VMM over-subscription that in-flight // work is itself PCIe-fault-slowed, so folding it into `GLOBAL_VRAM_FREE_NS` // mis-attributes paging-stalled compute/copy as "free" time (#1295). Timed -// separately so `vram_free_ns` measures only unmap/release. +// separately so the stream drain is excluded from `vram_free_ns`. +// +// CAUTION (2026-08-19 reconciliation, docs/benchmarks/ +// 2026-08-19-vram-free-attribution-reconciliation.md): even with the drain +// split out, `GLOBAL_VRAM_FREE_NS` is NOT "driver freeing time". It wraps the +// whole free code path -- the `cuMemUnmap`/`cuMemRelease` driver calls AND the +// per-granule Rust bookkeeping in `decommit_allocation_range`/`deallocate_span` +// -- and the bookkeeping dominates: on an RTX 4060 weight-lending run the driver +// `cuMemUnmap` was a stable ~16.9 ms/call (~101 ms total over 6 releases) while +// `vram_free_ms` swung 82 ms <-> 2450 ms (~30x) on byte-identical deterministic +// work. A metric that varies 30x while the work is constant is measuring +// variable non-driver time (bookkeeping and/or lock/blocking), not freeing. +// To isolate the driver cost, time the `cuMemUnmap` sites directly (see the +// reconciliation doc's split instrumentation); do not read `vram_free_ms` as a +// driver-freeing figure. static GLOBAL_VRAM_FREE_SYNC_NS: AtomicU64 = AtomicU64::new(0); // Process-lifetime high-water gauge. Resetting activity counters must not write // it: a concurrent page-in could otherwise be overwritten with a stale value. @@ -177,6 +191,13 @@ pub struct GlobalOffloadStats { pub materialize_fallback_calls: u64, pub htod_bytes: u64, pub vram_alloc_ns: u64, + /// Weight-page free code path: the `cuMemUnmap`/`cuMemRelease` driver calls + /// plus the per-granule Rust bookkeeping in `decommit_allocation_range`/ + /// `deallocate_span`. NOT a driver-freeing figure -- the bookkeeping + /// dominates and this counter swings ~30x on byte-identical work while the + /// driver `cuMemUnmap` stays ~17 ms/call (see `GLOBAL_VRAM_FREE_NS` and the + /// 2026-08-19 reconciliation doc). Excludes the pre-free stream drain, which + /// is timed into [`Self::vram_free_sync_ns`]. pub vram_free_ns: u64, /// Pre-free stream drain time (`cuStreamSynchronize`) taken on the Drop-path /// eviction before unmapping. Split out of [`Self::vram_free_ns`] so the From e211f03e7a27ba8df29ea9d1b4fd68aa34a41681 Mon Sep 17 00:00:00 2001 From: justinchuby <223556219+Copilot@users.noreply.github.com> Date: Wed, 19 Aug 2026 02:18:43 -0700 Subject: [PATCH 2/2] fix(cpu-ep): gate x86_64-only prefill fan-out symbols for non-x86 targets MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The aarch64 clippy gate is red on main (#1415): WIDE_PREFILL_MACS, prefill_fan_out and prefill_column_grain are only reachable from the x86_64 prefill route, so on other architectures they are genuinely dead and `-D warnings` fails the build. Adds `#[cfg(target_arch = "x86_64")]` to the three symbols and to the six tests that exercise them, matching the treatment their neighbours already have. Verified both sides, because gating dead code is only half the job: - aarch64-pc-windows-msvc: reproduced the three "never used" diagnostics on unmodified main, then `cargo clippy --target aarch64-pc-windows-msvc --all-targets -- -D warnings` exits 0 with this change. (The issue cites aarch64-unknown-linux-gnu; that target's std is not installed here, but the symbols are dead on any non-x86_64 target so the aarch64 Windows target reproduces and clears the same diagnostics.) - x86_64: `cargo clippy --all-targets -- -D warnings` exits 0, and the six gated tests still run — 26 passed / 0 failed / 2 ignored. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: d60eb808-7cc6-4abc-b48d-2a6dd3841624 --- crates/onnx-runtime-ep-cpu/src/kernels/matmul_nbits.rs | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/crates/onnx-runtime-ep-cpu/src/kernels/matmul_nbits.rs b/crates/onnx-runtime-ep-cpu/src/kernels/matmul_nbits.rs index 817c05383b..6127327461 100644 --- a/crates/onnx-runtime-ep-cpu/src/kernels/matmul_nbits.rs +++ b/crates/onnx-runtime-ep-cpu/src/kernels/matmul_nbits.rs @@ -392,6 +392,7 @@ enum PrefillFanOut { /// /// Park latency is what makes the wide path safe above the threshold: 226 us of /// worst-case wake-up is 0.25% of a 90 ms fan-out, and 45% of a 0.5 ms one. +#[cfg(target_arch = "x86_64")] const WIDE_PREFILL_MACS: usize = 1 << 29; /// Picks the prefill fan-out executor for `macs` of work, given the task @@ -399,6 +400,7 @@ const WIDE_PREFILL_MACS: usize = 1 << 29; /// /// Split out as a pure function so the policy is testable without a machine /// that has SMT, and so the threshold has one place to be wrong. +#[cfg(target_arch = "x86_64")] fn prefill_fan_out(macs: usize, lanes: usize, wide: usize) -> PrefillFanOut { // Nothing to win from the wide path when it is not actually wider; prefer // the runtime's cheaper dispatch. @@ -538,6 +540,7 @@ const MIN_PREFILL_TASK_MACS: usize = 1 << 19; /// /// Returns a *floor* the task runtime applies to its own partition; the runtime /// still uses a larger grain when there are more columns than workers. +#[cfg(target_arch = "x86_64")] fn prefill_column_grain(m: usize, k: usize, n: usize) -> usize { let macs_per_column = m.saturating_mul(k); if macs_per_column == 0 { @@ -17298,6 +17301,7 @@ mod tests { /// A prefill small enough that the task runtime's ~5 us dispatch dominates /// stays on the task runtime, whatever the widths look like. #[test] + #[cfg(target_arch = "x86_64")] fn small_prefill_work_stays_on_the_task_runtime() { assert_eq!( prefill_fan_out(WIDE_PREFILL_MACS - 1, 16, 32), @@ -17310,6 +17314,7 @@ mod tests { /// on work long enough for a 226 us wake-up to be noise, so take the wide /// path. #[test] + #[cfg(target_arch = "x86_64")] fn large_prefill_work_takes_the_wide_fan_out() { assert_eq!( prefill_fan_out(WIDE_PREFILL_MACS, 16, 32), @@ -17325,6 +17330,7 @@ mod tests { /// have. When it has them -- no SMT, an explicit task-thread budget, a /// narrow cpuset -- the cheaper dispatch wins unconditionally. #[test] + #[cfg(target_arch = "x86_64")] fn the_wide_fan_out_is_not_taken_when_it_is_not_wider() { for wide in 1..=16 { assert_eq!( @@ -17367,6 +17373,7 @@ mod tests { /// The native fan-out's grain is a floor in *output columns*, so it must /// never exceed the column count nor drop below one. #[test] + #[cfg(target_arch = "x86_64")] fn prefill_column_grain_stays_within_the_column_count() { for &(m, k, n) in &[ (1usize, 1usize, 1usize), @@ -17383,6 +17390,7 @@ mod tests { /// A column that already carries enough arithmetic is handed out one per /// task; thinner columns get batched until they clear the floor. #[test] + #[cfg(target_arch = "x86_64")] fn prefill_column_grain_batches_only_undersized_columns() { // 8 x 1024 = 8 Ki MACs a column, so batch 64 of them to clear 512 Ki. assert_eq!(prefill_column_grain(8, 1024, 3072), 64); @@ -17394,6 +17402,7 @@ mod tests { /// A degenerate shape must not divide by zero or ask for a zero grain. #[test] + #[cfg(target_arch = "x86_64")] fn prefill_column_grain_survives_a_zero_sized_problem() { assert_eq!(prefill_column_grain(0, 1024, 8), 8); assert_eq!(prefill_column_grain(8, 0, 8), 8);