diff --git a/cmd/waired-agent/hardware_summary_test.go b/cmd/waired-agent/hardware_summary_test.go index 613e8b44f..6183c5b6c 100644 --- a/cmd/waired-agent/hardware_summary_test.go +++ b/cmd/waired-agent/hardware_summary_test.go @@ -239,7 +239,19 @@ func TestHardwareSummaryFor_AppleSiliconBudgetSurvivesTheWireAdapter(t *testing. // which is the whole point of recording it as a debt. RAMAvailableMeasuredAt // (#699) was the most recent, held here between the proto PR and this one, // and it dated the very field that sat here before it. Empty again. -var notPublishedByAgent = map[string]bool{} +var notPublishedByAgent = map[string]bool{ + // waired-agent#69, held here for exactly one PR. The contract had to + // land alone (docs/decisions/20260719/0000-concurrent-proto-development.md + // §2), and hardware.GPU has no free-VRAM field to publish from until + // the per-OS reader lands with it. Until then every device reports 0, + // which hostfit reads as "no free reading" and answers with the + // total — so the budget is unchanged rather than wrong. + // + // The reader PR deletes this line. If it is still here, the contract + // is published with no producer, which is the #251 shape this map + // exists to make visible rather than to excuse. + "HardwareGPUSummary.VRAMFreeMB": true, +} // TestHardwareSummaryFor_PublishesEveryWireField guards the bug class // rather than the three fields: a field added to the broadcast summary diff --git a/docs/decisions/20260727/1830-ollama-vram-pool-nvidia-only.md b/docs/decisions/20260727/1830-ollama-vram-pool-nvidia-only.md index fa458f598..fd8a7f4bc 100644 --- a/docs/decisions/20260727/1830-ollama-vram-pool-nvidia-only.md +++ b/docs/decisions/20260727/1830-ollama-vram-pool-nvidia-only.md @@ -1,11 +1,26 @@ --- status: accepted +superseded_by: + - docs/decisions/20260813/1120-ollama-budget-sized-on-free-vram.md --- # ollama の VRAM プールは proto 側で導出し、当面 NVIDIA に限る (20260727 18:30) ## Status -Accepted +Accepted(一部置換) + +**§4「安全性は数値の精度ではなく accessor の床で担保する」だけ**が +`docs/decisions/20260813/1120-ollama-budget-sized-on-free-vram.md` に置き換わった。 +床という構造は残り、**何に対する床か**が「単体デバイスの total」から +「単体デバイスの free(測れたとき)」に移っている。§4 の「今日の挙動より +悪くならない」保証はそこで意図的に失効する — この決定が直していた誤りは +複数枚での**過小**評価だけであり、#69 が直すのは 1 枚での**過大**評価で、 +total に置いた床は後者を丸ごと飲み込むため。 + +§1(プールを proto 側で導出し wire に新フィールドを足さない)、§2(合算対象は +当面 NVIDIA のみ)、§3(控除はデバイス増分あたり)は**そのまま有効**。 +Consequences が「#69 の担当範囲として残す。混ぜると両方おかしくなる」と +指定した分離も、そのとおりに守られている(de-rate は合算の前・デバイスごと)。 ## Context diff --git a/docs/decisions/20260813/1120-ollama-budget-sized-on-free-vram.md b/docs/decisions/20260813/1120-ollama-budget-sized-on-free-vram.md new file mode 100644 index 000000000..3f57a9c6c --- /dev/null +++ b/docs/decisions/20260813/1120-ollama-budget-sized-on-free-vram.md @@ -0,0 +1,104 @@ +--- +status: accepted +supersedes: + - docs/decisions/20260727/1830-ollama-vram-pool-nvidia-only.md +--- + +# ollama の VRAM 予算は total ではなく free で測る。床の基準も free に移す (20260813 11:20) + +## Status +Accepted + +`docs/decisions/20260727/1830-ollama-vram-pool-nvidia-only.md` の **§4(安全性は +accessor の床で担保する)だけ**を置き換える。同決定の §1〜§3(プールを proto 側で +導出する / 合算対象は当面 NVIDIA のみ / 控除はデバイス増分あたり)はそのまま有効で、 +1830 は `accepted` のまま残る。 + +## Context + +`hostfit` はこれまで VRAM 予算を **total** で測っていた。一方 ollama の +スケジューラが見るのは **free** である。`docs/knowledges/20260727/1830-ollama-multi-gpu-placement.md` +が固定エンジン(0.31.1)のソースを読んで記録している既知のずれ、逐語: + +> スケジューラが見るのは **free** メモリ、こちらが合算するのは **total**。1 枚の +> ときからある差だが、枚数分だけ拡大する(#69)。しかも +> `bestGPUGroupByAvailableMemory` には `bestSingleGPUFit` のような 80% の割引がない。 + +害は具体的である。ディスプレイを駆動している 8 GB カードは、コンポジタと他プロセスが +数百 MB〜数 GB を保持したまま「8 GB 使える」と値付けされる。モデルとコンテキストは +その楽観的な数字で選ばれ、実際にロードすると spill し、#621 の post-load verify が +気づいてコンテキストを縮め、エンジンを 1 回再起動する。**選定は直らない** — verify が +直せるのは窓であってモデルの選択ではない。 + +1830 §4 はこの床を置いた: + +> `OllamaVRAMBudgetMB()` は (a) unified ホストでは `UsableVRAMMB` を必ず優先し、 +> (b) **単体デバイスの値を下回らない**。……これがあるので `OllamaVRAMPoolMB` が +> どう間違っていても今日の挙動より悪くならない。 + +当時それは正しかった。**当時直していた誤りが「複数枚での過小評価」だけだったから**である。 +床は「プール計算がどう間違っても 1 枚分は必ず確保される」ことを保証し、 +waired#942 の逆向き(動くものを断る)に転ぶのを構造的に防いでいた。 + +**#69 が直す誤りは向きが逆で、同じ床がそれを丸ごと飲み込む。** 1 枚のホストでは +`VRAMPoolMB == 0` なので予算は `EffectiveVRAMMB`(= total)に落ち、free は式に +現れない。複数枚でも、free 合算が 1 枚の total を下回った瞬間にクランプされる。 +つまり **床を total に置いたまま free を導入しても、#69 の主症例には何も起きない。** + +## Decision + +1. **デバイス単位の貸出可能量を `free`(測れたとき)とする。** `Device.lendableMB()` は + `VRAMFreeMB > 0 && VRAMFreeMB < VRAMTotalMB` のときだけ free を返し、それ以外は + total。`OllamaVRAMPoolMB` はこれを合算する。**de-rate は合算の前・デバイスごと**に + 効く(1830 の Consequences が #69 の担当範囲として明示的に残した形)。 + +2. **床の基準を「1 枚の total」から「1 枚の free(測れたとき)」に移す。** + `Host.ollamaSingleDeviceMB()` を新設し、`OllamaVRAMBudgetMB` はそれを床に使う。 + 床という**構造は残す** — 変わるのは何に対する床かだけである。 + +3. **`EffectiveVRAMMB` は動かさない。** 1830 が「`min_vram_mb`、エンジン選択、vLLM の + TP=1 フォールバックは全て『1 枚の値』として書かれている」と述べたとおりで、 + その判断は今も有効。free 対応は ollama 予算の経路にだけ入れる。 + +4. **unified-memory ホストは対象外。** `UsableVRAMMB` が既に共有プールから GPU が + wire down できる正直な上限であり、出荷済みのどの検出器も UMA デバイスの free を + 報告しない。改善する余地がなく、フォールバックは推測になる。 + +5. **測定は 1 回きり、自分のエンジンが weights を持つ前。** ハードウェアプロファイルは + TTL で再サンプルされるので、ロード後に取った free は**自分の重みを除外する** — + 再 tune のたびに空きが減って縮み続ける螺旋になり、ホストが自分の serve している + モデルを自分に課すことになる。`RAMAvailableGB` が #568 で同じ危険を同じ言葉で + 名指しし、同じ答え(1 回測って永続化・ライブでは読まない)を採っている。それに倣う。 + +6. **`0` は「free 未測定」であって「空きゼロ」ではない。** total にフォールバックする。 + free を報告しないドライバも、フィールドを知らない旧 agent も、ここに着地して + 今日の予算を保つ。これが 5 と合わせて、**測っていないホストを de-rate しない** + ことを構造的に保証する。 + +## Consequences + +- **1830 §4 の「今日の挙動より悪くならない」保証は失効する。** それが目的である。 + 予算は測定された free の分だけ**下がりうる**。代わりの保証は 6 の + 「測っていなければ動かさない」— 悪化しうるのは、ドライバが実際に空きを報告した + ホストに限られる。 +- `TestOllamaBudgetNeverShrinksTheHost`(`OllamaVRAMBudgetMB() >= EffectiveVRAMMB()` の + 全数掃引)は**意図的に反転する**。新しい不変条件は + `OllamaVRAMBudgetMB() >= ollamaSingleDeviceMB()` かつ + `ollamaSingleDeviceMB() <= EffectiveVRAMMB()`。 +- 過小評価に転ぶ経路が新しく開く(測定時に他プロセスが一時的に VRAM を掴んでいた + 場合)。5 の「静かな瞬間に 1 回測る」がその窓を狭める唯一の手段であり、 + #568 が RAM について既に受け入れたのと同じトレードオフである。 +- エンジンとの整合は改善する。`bestSingleGPUFit` は free の 80% で判定するので、 + free で測る予算は依然としてエンジンより**楽観的**なまま — つまりこの変更は + エンジンより厳しくなる方向には行き過ぎない。 +- 1830 の Consequences が挙げた残件のうち、`#266`(カード別帯域テーブル)と + 「`EffectiveVRAMMB` が列挙順の `GPUs[0]` を信じている件」(#264 調査項目 6)は + **未着手のまま**。この決定はどちらにも触れない。 + +## Refs +- https://github.com/waired-ai/waired-agent/issues/69 +- https://github.com/waired-ai/waired-agent/issues/264 +- https://github.com/waired-ai/waired-agent/issues/568 — 同型の「1 回測って永続化」 +- `docs/decisions/20260727/1830-ollama-vram-pool-nvidia-only.md`(§4 を置換) +- `docs/decisions/20260727/1240-host-fit-single-source-proto-hostfit.md` +- `docs/knowledges/20260727/1830-ollama-multi-gpu-placement.md` diff --git a/proto/hostfit/hostfit.go b/proto/hostfit/hostfit.go index 301ac3c70..00a48f496 100644 --- a/proto/hostfit/hostfit.go +++ b/proto/hostfit/hostfit.go @@ -303,6 +303,22 @@ type Host struct { // asks a field added to a published struct to say so explicitly. VRAMPoolMB int `json:"-"` + // VRAMAvailable0MB is the first GPU's free VRAM — the single-device + // mirror of VRAMPoolMB, and the only way the free reading reaches a + // SINGLE-GPU host, which is the shape waired-agent#69 actually + // reported (an 8 GB card also driving the display). A pooled host + // gets its de-rate inside VRAMPoolMB; a one-card host has no pool, + // so without this field it would keep sizing against the total. + // + // 0 means "no free reading" and leaves the total in place. Read only + // through OllamaVRAMBudgetMB, never directly — EffectiveVRAMMB stays + // the raw single-device figure that min_vram_mb, engine selection + // and vLLM's TP=1 fallback were authored against, for the same + // reason VRAMPoolMB may not widen it. + // + // json:"-" for the reason above: an input, not a payload. + VRAMAvailable0MB int `json:"-"` + // MemoryBandwidthSpecGBs is the published PEAK read bandwidth of the // pool the weights are read from, in GB/s. 0 means "unknown", and // that is a case this package must keep working for rather than an @@ -403,6 +419,45 @@ type Device struct { // the AMD Windows registry fallback reports devices this way — and // such a device contributes nothing to a pool. VRAMTotalMB int + + // VRAMAvailableMB is how much of VRAMTotalMB the driver reported + // free, measured once while no engine of ours held weights. 0 means + // "no free reading" and falls back to the total, which is what every + // producer sent before the field existed + // (signer.HardwareGPUSummary.VRAMFreeMB carries the discipline and + // the reason, waired-agent#69). + // + // Named "available" rather than "free" — matching Host.RAMAvailableGB + // for the same quantity — partly to keep scripts/ci/protoconsumer + // working. That guard matches producers by field NAME, so a Device + // field spelled VRAMFreeMB and assigned right here in + // FromHardwareSummary would read as a proto-internal producer for the + // WIRE field of that name, and the real producer debt would never + // become visible in its table. The same consideration named + // LocalModelChoiceAt (waired-agent#647); the driver's own word stays + // on the wire, where nvidia-smi's memory.free is what it reports. + // + // json:"-" for the reason Host.VRAMPoolMB carries it: Device is an + // INPUT both adapters build, never a payload — the wire shape is + // signer.HardwareGPUSummary — and the additive-only guard asks a + // field added to a published struct to say so explicitly. Its + // untagged siblings predate the guard's baseline. + VRAMAvailableMB int `json:"-"` +} + +// lendableMB is what this device can actually lend an engine: its free +// reading where the driver gave one, its total otherwise. +// +// The fallback is the whole safety argument. A device whose driver will +// not report free memory, and a producer that predates the field, both +// arrive here as 0 and get the total — so this rule can only ever +// de-rate a device the reader actually measured, never one it guessed +// at. +func (d Device) lendableMB() int { + if d.VRAMAvailableMB > 0 && d.VRAMAvailableMB < d.VRAMTotalMB { + return d.VRAMAvailableMB + } + return d.VRAMTotalMB } // OllamaVRAMPoolMB is the VRAM ollama may pool across devices, or 0 @@ -454,7 +509,15 @@ func OllamaVRAMPoolMB(devs []Device) int { continue } n++ - sum += d.VRAMTotalMB + // Free where it was measured, total otherwise. Summing free is + // what the engine itself does — availableMemoryForLoad sums + // gpu.FreeMemory — so the pool now answers the question the + // scheduler asks rather than an optimistic neighbour of it + // (waired-agent#69). The de-rate is applied per device, BEFORE + // the sum, which is what #264's decision record asks for: the + // total−free gap is per-device, so summing totals accumulates + // it once per card. + sum += d.lendableMB() } if n < 2 { // Nothing to pool. Reported as "unknown" rather than as the @@ -487,14 +550,21 @@ func FromHardwareSummary(hw *signer.HardwareSummary) Host { } if len(hw.GPUs) > 0 { h.VRAM0MB = hw.GPUs[0].VRAMTotalMB + h.VRAMAvailable0MB = hw.GPUs[0].VRAMFreeMB } // Every GPU has always been on the wire; only this adapter and its - // agent-side twin threw the rest away. Nothing new has to be - // published for the pool, so the fix reaches every already-deployed - // agent the moment the control plane bumps its proto tag. + // agent-side twin threw the rest away. Nothing new had to be + // published for the pool, so that fix reached every already-deployed + // agent the moment the control plane bumped its proto tag. The free + // reading is the exception: it is a new field, so it arrives only + // from an agent new enough to measure it, and 0 keeps the total. devs := make([]Device, 0, len(hw.GPUs)) for _, g := range hw.GPUs { - devs = append(devs, Device{Vendor: g.Vendor, VRAMTotalMB: g.VRAMTotalMB}) + devs = append(devs, Device{ + Vendor: g.Vendor, + VRAMTotalMB: g.VRAMTotalMB, + VRAMAvailableMB: g.VRAMFreeMB, + }) } h.VRAMPoolMB = OllamaVRAMPoolMB(devs) return h @@ -529,20 +599,52 @@ func (h Host) EffectiveVRAMMB() int { // the GPU can wire down, so it keeps winning. // // The aggregate may never come in BELOW the single-device figure. That -// is the mirror of router.VLLMVRAMBudgetMB's own floor, and here it is -// what makes this change structurally unable to regress: the worst a -// wrong pool can do is leave today's behaviour in place. It is also why +// is the mirror of router.VLLMVRAMBudgetMB's own floor. It is also why // no "floor at the largest device" clause is needed — a host whose // GPUs[0] is its small card gets the pool, which already exceeds the // large one. Whether EffectiveVRAMMB itself should rank devices rather // than trust enumeration order is a separate question, deliberately not // answered here (waired-ai/waired-agent#264 item 6). +// +// What the floor is measured AGAINST changed with waired-agent#69: it +// is the single device's FREE figure where one was measured, not its +// total. docs/decisions/20260813/1120-ollama-budget-sized-on-free-vram.md +// records why, and revises §4 of the pool decision that set the earlier +// floor. The short version is that a floor at the total made the budget +// structurally unable to shrink, which was the point while the only +// error being corrected was an UNDER-count across cards — and is +// exactly wrong once the error being corrected is an OVER-count on one. func (h Host) OllamaVRAMBudgetMB() int { + single := h.ollamaSingleDeviceMB() + if h.UnifiedMemory || h.VRAMPoolMB <= single { + return single + } + return h.VRAMPoolMB +} + +// ollamaSingleDeviceMB is EffectiveVRAMMB as the OLLAMA path must read +// it: the same single-device budget, de-rated to what the driver +// reported free. +// +// Separate from EffectiveVRAMMB rather than folded into it, because the +// pool decision already settled that widening or narrowing THAT figure +// moves min_vram_mb, engine selection and vLLM's TP=1 fallback, all of +// which were authored against a whole card. This one moves only the +// ollama budget, which is the only consumer waired-agent#69 is about. +// +// A unified-memory host is left alone: UsableVRAMMB is already the +// honest bound on what its GPU can wire down from a shared pool, and no +// shipped detector reports a free figure for one, so there is nothing +// here to improve and a fallback to guess at. +func (h Host) ollamaSingleDeviceMB() int { eff := h.EffectiveVRAMMB() - if h.UnifiedMemory || h.VRAMPoolMB <= eff { + if h.UnifiedMemory { return eff } - return h.VRAMPoolMB + if h.VRAMAvailable0MB > 0 && h.VRAMAvailable0MB < eff { + return h.VRAMAvailable0MB + } + return eff } // HasGPU reports whether the host has any GPU-addressable memory at diff --git a/proto/hostfit/hostfit_test.go b/proto/hostfit/hostfit_test.go index 21e8049e5..1da4932d1 100644 --- a/proto/hostfit/hostfit_test.go +++ b/proto/hostfit/hostfit_test.go @@ -68,6 +68,21 @@ const ( `{"model":"NVIDIA GeForce RTX 4090","vram_total_mb":24564,"compute_cap":"8.9","vendor":"nvidia"},` + `{"model":"AMD Radeon RX 7900 XTX","vram_total_mb":24576,"vendor":"amd"}],` + `"ram_total_gb":128}` + // The waired-agent#69 host, and the reason the free reading exists: a + // low-VRAM desktop card that is ALSO driving the display, so ~2 GB of + // its 8 are already spoken for before any model loads. Sized against + // the total it is offered more than it can hold, spills, and #621's + // post-load verify has to shrink the window and restart the engine. + wireRTX3060TiBusy = `{"gpus":[{"model":"NVIDIA GeForce RTX 3060 Ti","vram_total_mb":8192,` + + `"vram_free_mb":6144,"compute_cap":"8.6","vendor":"nvidia"}],"ram_total_gb":32}` + // The same two cards as wireDual4090, from an agent new enough to + // measure free memory. Each card is holding ~1.5 GB, so the pool is + // 3 GB smaller than the totals suggest — the per-device gap, once per + // card, which is what #264's record predicted would accumulate. + wireDual4090Busy = `{"gpus":[` + + `{"model":"NVIDIA GeForce RTX 4090","vram_total_mb":24564,"vram_free_mb":23028,"compute_cap":"8.9","vendor":"nvidia"},` + + `{"model":"NVIDIA GeForce RTX 4090","vram_total_mb":24564,"vram_free_mb":23028,"compute_cap":"8.9","vendor":"nvidia"}],` + + `"ram_total_gb":128}` ) func hostFromWire(t *testing.T, payload string) hostfit.Host { @@ -194,6 +209,9 @@ func TestEffectiveVRAMMB(t *testing.T) { func TestOllamaVRAMPoolMB(t *testing.T) { nv := func(mb int) hostfit.Device { return hostfit.Device{Vendor: "nvidia", VRAMTotalMB: mb} } amd := func(mb int) hostfit.Device { return hostfit.Device{Vendor: "amd", VRAMTotalMB: mb} } + nvFree := func(total, free int) hostfit.Device { + return hostfit.Device{Vendor: "nvidia", VRAMTotalMB: total, VRAMAvailableMB: free} + } for _, tc := range []struct { name string @@ -241,6 +259,37 @@ func TestOllamaVRAMPoolMB(t *testing.T) { []hostfit.Device{{Vendor: "apple", VRAMTotalMB: 24576}}, 0, }, + { + // waired-agent#69: the engine sums FreeMemory, so the pool + // has to as well. Two cards holding 1.5 GB each come in + // 3 GB under the totals — the per-device gap, once per card. + "measured cards pool their free memory, not their totals", + []hostfit.Device{nvFree(24564, 23028), nvFree(24564, 23028)}, + 23028*2 - 1024, + }, + { + // The de-rate is per device and BEFORE the sum, which is what + // #264's record asked #69 to do rather than compensating on + // the pooling side. + "a card with no free reading contributes its total", + []hostfit.Device{nvFree(24564, 23028), nv(24564)}, + 23028 + 24564 - 1024, + }, + { + // 0 is "not measured", never "nothing free". A fleet that has + // not updated must keep today's pool exactly. + "no free readings anywhere reproduce the old pool", + []hostfit.Device{nvFree(24564, 0), nvFree(24564, 0)}, + 24564*2 - 1024, + }, + { + // A driver that reports free >= total is not telling us + // anything the total did not, and trusting it would let a + // bogus reading INFLATE a host. Only a de-rate is honoured. + "a free reading at or above the total is ignored", + []hostfit.Device{nvFree(24564, 30000), nvFree(24564, 24564)}, + 24564*2 - 1024, + }, } { t.Run(tc.name, func(t *testing.T) { if got := hostfit.OllamaVRAMPoolMB(tc.devs); got != tc.want { @@ -279,6 +328,42 @@ func TestOllamaVRAMBudgetMB(t *testing.T) { hostfit.Host{GPUCount: 2, VRAM0MB: 24564, VRAMPoolMB: 20000}, 24564, }, + { + // waired-agent#69, the reported shape: ONE card, so there is + // no pool to carry the de-rate, and without VRAMAvailable0MB the + // free reading would never reach the budget at all. + "a single measured card is sized on what is free", + hostFromWire(t, wireRTX3060TiBusy), + 6144, + }, + { + "two measured cards spend the measured pool", + hostFromWire(t, wireDual4090Busy), + 23028*2 - 1024, + }, + { + // The floor moved from the device's total to its free figure, + // so a degenerate pool still cannot shrink the host BELOW + // what the one card was measured to have. + "a pool below the measured single device is still ignored", + hostfit.Host{GPUCount: 2, VRAM0MB: 24564, VRAMAvailable0MB: 20000, VRAMPoolMB: 12000}, + 20000, + }, + { + // No shipped detector reports free memory for a unified part, + // and UsableVRAMMB is already the honest bound on what its + // GPU can wire down. A stray reading must not move it. + "unified memory ignores a free reading", + hostfit.Host{UnifiedMemory: true, GPUCount: 1, UsableVRAMMB: 18432, VRAM0MB: 24576, VRAMAvailable0MB: 4096}, + 18432, + }, + { + // The whole fleet before this field, and every driver that + // will not answer: unchanged. + "an unmeasured card keeps its total", + hostfit.Host{GPUCount: 1, VRAM0MB: 24564, VRAMAvailable0MB: 0}, + 24564, + }, } { t.Run(tc.name, func(t *testing.T) { if got := tc.host.OllamaVRAMBudgetMB(); got != tc.want { @@ -288,19 +373,22 @@ func TestOllamaVRAMBudgetMB(t *testing.T) { } } -// TestOllamaBudgetNeverShrinksTheHost is the anti-waired#942 invariant, -// and the reason this change is safe to make without a per-host -// measurement behind it. +// TestOllamaBudgetNeverShrinksAHostItDidNotMeasure is what survives of +// the old TestOllamaBudgetNeverShrinksTheHost, and it is the half that +// still holds. // -// #264 is an UNDER-count: a host is refused a model it runs. The -// opposite error — over-counting, and offering a model that spills — is -// the failure waired#942 was, and it is the one this package must never -// reintroduce while fixing the first. The accessor's floor makes that -// structural rather than a matter of getting the pool arithmetic right: -// whatever OllamaVRAMPoolMB returns, wrong or right, the budget cannot -// come in under today's figure, so no producer error can shrink a -// host's catalog. -func TestOllamaBudgetNeverShrinksTheHost(t *testing.T) { +// PRODUCT CONTRACT — docs/decisions/20260813/1120-ollama-budget-sized-on-free-vram.md +// §Decision 6. That decision deliberately gave up the blanket "the +// budget can never come in below today's figure" guarantee: sizing on +// free memory means the budget CAN shrink, which is the entire point of +// waired-agent#69. What it did not give up is the guarantee for a host +// nobody measured — a driver that will not report free memory, and +// every agent that predates the field, both arrive with 0 and must keep +// exactly today's arithmetic. +// +// So the sweep is unchanged except that VRAMAvailable0MB stays 0 throughout. +// If a future change lets an unmeasured host be de-rated, this fails. +func TestOllamaBudgetNeverShrinksAHostItDidNotMeasure(t *testing.T) { for _, unified := range []bool{false, true} { for _, usable := range []int{0, 8192, 18432} { for _, vram0 := range []int{0, 4096, 24564, 49152} { @@ -310,8 +398,9 @@ func TestOllamaBudgetNeverShrinksTheHost(t *testing.T) { UsableVRAMMB: usable, VRAM0MB: vram0, VRAMPoolMB: pool, } if got, floor := h.OllamaVRAMBudgetMB(), h.EffectiveVRAMMB(); got < floor { - t.Fatalf("%+v: ollama budget %d is BELOW the single-device figure %d — "+ - "a pool must never take a model away from a host that ran it", + t.Fatalf("%+v: ollama budget %d is BELOW the single-device figure %d "+ + "on a host with no free reading — an unmeasured host must keep "+ + "today's budget, not be de-rated on a guess", h, got, floor) } } @@ -320,6 +409,45 @@ func TestOllamaBudgetNeverShrinksTheHost(t *testing.T) { } } +// TestOllamaBudgetNeverFallsBelowWhatWasMeasured is the replacement +// invariant for the measured case. +// +// PRODUCT CONTRACT — the same decision, §Decision 2. The floor did not +// go away; it changed what it is measured against. A budget may come in +// under the card's TOTAL — that is the de-rate #69 asks for — but it +// may never come in under what the driver actually reported free, +// because nothing in this package knows anything more pessimistic than +// the measurement, and inventing a further haircut here would be the +// waired#942 direction (refusing a model the host runs) with no +// evidence behind it. +func TestOllamaBudgetNeverFallsBelowWhatWasMeasured(t *testing.T) { + for _, unified := range []bool{false, true} { + for _, usable := range []int{0, 8192, 18432} { + for _, vram0 := range []int{0, 4096, 24564, 49152} { + for _, free := range []int{0, 1, 4096, 20000, 49152} { + for _, pool := range []int{0, 1, 8192, 24564, 98304} { + h := hostfit.Host{ + GPUCount: 2, UnifiedMemory: unified, + UsableVRAMMB: usable, VRAM0MB: vram0, + VRAMAvailable0MB: free, VRAMPoolMB: pool, + } + floor := h.EffectiveVRAMMB() + if free > 0 && free < floor { + floor = free + } + if got := h.OllamaVRAMBudgetMB(); got < floor { + t.Fatalf("%+v: ollama budget %d is below %d, the lesser of the "+ + "single-device figure and the measured free reading — "+ + "the budget may de-rate to what was measured, never past it", + h, got, floor) + } + } + } + } + } + } +} + // TestOllamaVRAMOverheadMB pins the three arms of the overhead model. // The discrete slope is what makes a 22.6 GB model fit a 24 GB card; // the old flat 4096 did not. diff --git a/proto/signer/capability.go b/proto/signer/capability.go index 67c670547..e5ae83f22 100644 --- a/proto/signer/capability.go +++ b/proto/signer/capability.go @@ -131,4 +131,23 @@ const ( // have to trust an agent to get that right: it strips both when v1 // is missing, whatever v2 says. CapabilityRAMAvailableV2 = "ram-available-v2" + + // CapabilityVRAMFreeV1 declares that this agent understands + // HardwareGPUSummary.VRAMFreeMB — the per-device free-VRAM + // measurement the ollama budget is sized against + // (waired-agent#69). + // + // It needs the gate for the reason RAMAvailableV1 did, and for the + // same structural reason: the field is agent-reported and rides the + // signed map on every PEER entry, so an agent that does not know it + // drops it on canonical re-marshal and fails verification. The CP + // strips it across the whole map for an undeclared poller, not just + // from Self. + // + // A reader that predates the field sees 0, which every consumer + // already treats as "no free reading" and answers by falling back to + // VRAMTotalMB — an old agent simply sizes the budget the way it does + // today. That is also what makes the field safe to publish before + // any producer fills it in. + CapabilityVRAMFreeV1 = "vram-free-v1" ) diff --git a/proto/signer/inference_state.go b/proto/signer/inference_state.go index 5ca2b65f1..d4dd1a29a 100644 --- a/proto/signer/inference_state.go +++ b/proto/signer/inference_state.go @@ -693,6 +693,33 @@ type HardwareGPUSummary struct { // VRAMTotalMB is the device's total VRAM in megabytes. VRAMTotalMB int `json:"vram_total_mb,omitempty"` + // VRAMFreeMB is how much of VRAMTotalMB the driver reported as free + // on this device, in megabytes. It exists because the engine sizes + // placement against free memory while this repo sized its budget + // against the total, so a card also driving a display was valued at + // more than it can lend (waired-agent#69). + // + // It is measured ONCE, while no engine or model of ours is resident, + // and persisted — never a live reading. That discipline is not + // optional: the hardware profile is re-sampled on a TTL, and a free + // figure taken after our own weights loaded would exclude them, so + // each re-tune would see less memory and shrink further. It is the + // same hazard RAMAvailableGB above names in the same words, and the + // same answer (waired-agent#568). + // + // 0 means "no free reading", never "no free VRAM": a consumer falls + // back to VRAMTotalMB, which is what a pre-addition agent sends and + // what a driver that will not report free memory leaves behind. So + // an unknown reading keeps today's budget rather than de-rating a + // host to nothing — the "judgement withheld when the budget is + // unknown" rule docs/decisions/20260728/0250-gpu-presence-from-driver-not-path.md + // settled for VRAM figures generally. + // + // Gated behind CapabilityVRAMFreeV1: agent-reported and riding every + // PEER entry, so the CP strips it across the whole map for a poller + // that has not declared it. + VRAMFreeMB int `json:"vram_free_mb,omitempty"` + // ComputeCap is the CUDA compute capability formatted as a // string (e.g. "8.9" for Ada Lovelace). Empty for non-CUDA. ComputeCap string `json:"compute_cap,omitempty"` diff --git a/proto/signer/inference_state_test.go b/proto/signer/inference_state_test.go index 0f1c54bfd..a285bf391 100644 --- a/proto/signer/inference_state_test.go +++ b/proto/signer/inference_state_test.go @@ -320,6 +320,94 @@ func TestHardwareSummary_RAMAvailable_CanonicalJSON(t *testing.T) { } } +// TestHardwareSummary_VRAMFree_CanonicalJSON is the byte-identity pin +// for the waired-agent#69 addition, and it carries the same weight the +// #568 pin above does: HardwareGPUSummary rides the signed NetworkMap +// inside HardwareSummary, so a shifted encoding churns the map for every +// peer on a rolling upgrade. +func TestHardwareSummary_VRAMFree_CanonicalJSON(t *testing.T) { + // Unmeasured: byte-for-byte the pre-#69 encoding. This is the whole + // fleet today, and every driver that will not answer. + unmeasured := HardwareSummary{ + GPUs: []HardwareGPUSummary{{ + Model: "NVIDIA GeForce RTX 4090", + VRAMTotalMB: 24564, + ComputeCap: "8.9", + Vendor: "nvidia", + }}, + RAMTotalGB: 64, + } + const wantUnmeasured = `{"gpus":[{"model":"NVIDIA GeForce RTX 4090","vram_total_mb":24564,` + + `"compute_cap":"8.9","vendor":"nvidia"}],"ram_total_gb":64}` + data, err := json.Marshal(&unmeasured) + if err != nil { + t.Fatalf("marshal unmeasured: %v", err) + } + if got := string(data); got != wantUnmeasured { + t.Errorf("an unmeasured device changed the encoding:\n got %s\nwant %s", got, wantUnmeasured) + } + + // Measured: the key sits between vram_total_mb and compute_cap, in + // struct-declaration order, next to the total it qualifies. + measured := HardwareSummary{ + GPUs: []HardwareGPUSummary{{ + Model: "NVIDIA GeForce RTX 3060 Ti", + VRAMTotalMB: 8192, + VRAMFreeMB: 6144, + ComputeCap: "8.6", + Vendor: "nvidia", + }}, + RAMTotalGB: 32, + } + const wantMeasured = `{"gpus":[{"model":"NVIDIA GeForce RTX 3060 Ti","vram_total_mb":8192,` + + `"vram_free_mb":6144,"compute_cap":"8.6","vendor":"nvidia"}],"ram_total_gb":32}` + data, err = json.Marshal(&measured) + if err != nil { + t.Fatalf("marshal measured: %v", err) + } + if got := string(data); got != wantMeasured { + t.Errorf("vram_free_mb encoding drifted:\n got %s\nwant %s", got, wantMeasured) + } + + // Round trip: the CP reads it off the stored push and must reach the + // same budget the agent reached. + var out HardwareSummary + if err := json.Unmarshal(data, &out); err != nil { + t.Fatalf("unmarshal: %v", err) + } + if !reflect.DeepEqual(&measured, &out) { + t.Errorf("round-trip mismatch\n in: %+v\nout: %+v", measured, out) + } + + // A pre-addition payload parses with the field zero, which every + // consumer treats as "no free reading" and answers with the total. + var pre HardwareSummary + if err := json.Unmarshal([]byte(wantUnmeasured), &pre); err != nil { + t.Fatalf("unmarshal pre-addition: %v", err) + } + if pre.GPUs[0].VRAMFreeMB != 0 { + t.Errorf("VRAMFreeMB = %d, want 0 on a pre-addition payload", pre.GPUs[0].VRAMFreeMB) + } +} + +// TestCapabilityVRAMFreeV1_WireValue pins the capability literal, and +// that it is distinct from its neighbours. The CP compares this exact +// string to decide whether to strip vram_free_mb across the whole served +// map, so a reword is a wire break rather than a rename — and an agent +// that receives the field without knowing it drops the key on canonical +// re-marshal and fails verification. +func TestCapabilityVRAMFreeV1_WireValue(t *testing.T) { + if CapabilityVRAMFreeV1 != "vram-free-v1" { + t.Fatalf("CapabilityVRAMFreeV1 = %q, want %q", + CapabilityVRAMFreeV1, "vram-free-v1") + } + for _, other := range []string{CapabilityRAMAvailableV1, CapabilityRAMAvailableV2} { + if CapabilityVRAMFreeV1 == other { + t.Fatalf("capability literals must stay distinct, both = %q", other) + } + } +} + // TestCapabilityRAMAvailableV1_WireValue pins the capability literal: // CP poll intake, distribution gate, and agent poller all compare this // exact string, so a reword is a wire-protocol break, not a rename. diff --git a/scripts/ci/protoconsumer/exemptions.go b/scripts/ci/protoconsumer/exemptions.go index fa9a5e59b..b60129a70 100644 --- a/scripts/ci/protoconsumer/exemptions.go +++ b/scripts/ci/protoconsumer/exemptions.go @@ -204,6 +204,16 @@ var producedInProto = []exemption{ "the retirement table, written only by proto/catalog/retired.go"}, {reflect.TypeFor[catalog.Retirement](), "SuccessorModelID", "the retirement table, written only by proto/catalog/retired.go"}, + // waired-agent#69's two hostfit halves. FromHardwareSummary — the + // control plane's adapter, in proto/ — writes both, so the category + // is accurate today. The agent's twin in internal/hardware/profiler.go + // cannot fill them until hardware.GPU carries the reading, and when it + // does this guard says "something under cmd/, internal/ now writes it + // — delete the entry", which is how the reader PR pays the debt. + {reflect.TypeFor[hostfit.Device](), "VRAMAvailableMB", + "waired-agent#69: FromHardwareSummary writes it; the agent-side HostFit() adapter follows with the reader"}, + {reflect.TypeFor[hostfit.Host](), "VRAMAvailable0MB", + "waired-agent#69: FromHardwareSummary writes it; the agent-side HostFit() adapter follows with the reader"}, } // producerPending: this repo owes the writer. Each entry names the issue @@ -249,7 +259,21 @@ var producedInProto = []exemption{ // PR paid it: hostMemoryMeasurement returns the persisted measured_at // beside the value it dates, and hardwareSummaryFor publishes it. Empty // again, and that is the point. -var producerPending = []exemption{} +var producerPending = []exemption{ + // waired-agent#69. The contract had to land alone + // (docs/decisions/20260719/0000-concurrent-proto-development.md §2), + // and there is nothing to publish from yet: hardware.GPU has no + // free-VRAM field until the per-OS reader lands with it, so unlike + // #568's RAMAvailableGB there is no same-named written field for the + // name-matching rule above to mistake for a producer. The debt is + // therefore visible here, which is what this table is for. + // + // Until it is paid every device sends 0, which hostfit reads as "no + // free reading" and answers with the total — the budget is unchanged + // rather than wrong, so the published contract is inert, not broken. + {reflect.TypeFor[signer.HardwareGPUSummary](), "VRAMFreeMB", + "waired-agent#69: the per-OS free-VRAM reader lands the producer; delete this entry in that PR"}, +} // exemption declares one proto field with no producer under cmd/ or // internal/. The struct is a reflect.Type rather than a string so the