Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 11 additions & 2 deletions cmd/waired-agent/engine_bootstrap.go
Original file line number Diff line number Diff line change
Expand Up @@ -348,8 +348,17 @@ func (p *agentInferenceProvider) bootstrapAfterEngineStart(ctx context.Context)
// next backend if it didn't, so the host never runs on CPU silently
// while a working GPU path exists. Conservative: an inconclusive
// probe keeps the preferred backend.
if p.bootPlan.backend.Probes() {
resolved := resolveBackendWithProbe(ctx, p.ollama, p.bootPlan.backend, p.ollama.BaseURL(), &http.Client{}, p.logger)
//
// No longer gated on Probes() (#70). A host with only one GPU backend
// cannot be moved to a better one, but it can still be MISLABELLED —
// a detected GPU that fails to engage kept reporting cuda / vulkan /
// metal while inference ran on the CPU. resolveBackendWithProbe now
// decides for itself what a plan's verdict may change: a restart
// where there is a fallback, the label alone where there is not.
// "" means the probe declined to decide (a provider with no boot
// plan); the seed from startInferenceSubsystem stands rather than
// being cleared.
if resolved := resolveBackendWithProbe(ctx, p.ollama, p.bootPlan.backend, p.ollama.BaseURL(), &http.Client{}, p.logger); resolved != "" {
p.ollama.SetResolvedBackend(resolved)
}
// #621: verify the exported serve tuning against the running engine
Expand Down
1 change: 1 addition & 0 deletions cmd/waired-agent/hardware_summary.go
Original file line number Diff line number Diff line change
Expand Up @@ -104,6 +104,7 @@ func hardwareSummaryFor(prof hardware.Profile) *signer.HardwareSummary {
summary.GPUs = append(summary.GPUs, signer.HardwareGPUSummary{
Model: g.Model,
VRAMTotalMB: g.VRAMTotalMB,
VRAMFreeMB: g.VRAMFreeMB,
ComputeCap: g.ComputeCap,
Vendor: g.Vendor,
})
Expand Down
26 changes: 10 additions & 16 deletions cmd/waired-agent/hardware_summary_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -239,19 +239,7 @@ func TestHardwareSummaryFor_AppleSiliconBudgetSurvivesTheWireAdapter(t *testing.
// which is the whole point of recording it as a debt. RAMAvailableMeasuredAt
// (#699) was the most recent, held here between the proto PR and this one,
// and it dated the very field that sat here before it. Empty again.
var notPublishedByAgent = map[string]bool{
// waired-agent#69, held here for exactly one PR. The contract had to
// land alone (docs/decisions/20260719/0000-concurrent-proto-development.md
// §2), and hardware.GPU has no free-VRAM field to publish from until
// the per-OS reader lands with it. Until then every device reports 0,
// which hostfit reads as "no free reading" and answers with the
// total — so the budget is unchanged rather than wrong.
//
// The reader PR deletes this line. If it is still here, the contract
// is published with no producer, which is the #251 shape this map
// exists to make visible rather than to excuse.
"HardwareGPUSummary.VRAMFreeMB": true,
}
var notPublishedByAgent = map[string]bool{}

// TestHardwareSummaryFor_PublishesEveryWireField guards the bug class
// rather than the three fields: a field added to the broadcast summary
Expand Down Expand Up @@ -282,9 +270,15 @@ func TestHardwareSummaryFor_PublishesEveryWireField(t *testing.T) {
// carry, so a zero in the output is the producer dropping it.
CarveOutVRAMMB: 49152,
GPUs: []hardware.GPU{{
Vendor: "apple",
Model: "Apple M3 Max",
VRAMTotalMB: 65536,
Vendor: "apple",
Model: "Apple M3 Max",
VRAMTotalMB: 65536,
// No Apple part reports a free figure either — the shipped
// readers are NVML and nvidia-smi. Populated for the same
// reason as the compute capability above: this fixture is
// every fact the profile CAN carry, so a zero in the output
// is the producer dropping it (waired-agent#69).
VRAMFreeMB: 61440,
ComputeCap: "8.9",
DriverVersion: "535.171.04",
UUID: "GPU-12345678",
Expand Down
41 changes: 38 additions & 3 deletions cmd/waired-agent/inference.go
Original file line number Diff line number Diff line change
Expand Up @@ -520,6 +520,12 @@ func startInferenceSubsystem(ctx context.Context, wg *sync.WaitGroup, logger *sl
preferencePath: deps.PreferencePath,
dlProgress: newDownloadProgress(),
ollamaUsable: func() bool { _, e := ollamaResolver(); return e == nil },
// The one rule, not the cached profile (#225). engineViable and
// setupEngineState already ask this way; this was the site that
// did not.
vllmUsable: func() bool {
return engineInstalledOnHost(runtime.GOOS, stateDir, catalog.RuntimeVLLM)
},
// engineChoice re-runs THIS boot's decision against the live host,
// so an adopt trigger asks the same rule rather than a snapshot
// taken before the engine existed (#339, the shape #304 gave the
Expand Down Expand Up @@ -1153,6 +1159,25 @@ type agentInferenceProvider struct {
// nil is treated as "not usable".
ollamaUsable func() bool

// vllmUsable is the same question for vllm, and exists because for a
// long time only ollama had one. The vllm arm of hasUsableEngine read
// hardware.Profile.Engines.VLLM.Installed instead, which made it the
// last engine-presence site not routed through engineInstalledOnHost
// (#225, the residual of the #179 class that PR #205 unified).
//
// The profile is no longer a PATH probe — #238 injects
// engineVersionOnHost into the daemon's profiler, so it does resolve
// the venv — but it is TTL-cached, and engine_resolve.go says in as
// many words why that is not good enough here: cached for 30 s, so it
// is still LATE for a fresh install, which is exactly what the wizard
// could not tolerate. A host whose venv appeared during setup could
// report no_engine for half a minute after it was usable.
//
// nil is treated as "not usable", and hasUsableEngine then falls back
// to the profile — the same shape ollamaUsable has, so a unit fixture
// that constructs the provider directly keeps working.
vllmUsable func() bool

// engineChoice answers "which engine would this host choose right now",
// by re-running the boot rule (chooseEngine) against the live state dir
// and hardware profile. ok=false means it could not answer — a strict
Expand Down Expand Up @@ -2160,7 +2185,7 @@ func subsystemState(f inferenceSubsystemFacts) string {
func (p *agentInferenceProvider) subsystemFacts(ctx context.Context, hw hardware.Profile, st catalog.State) inferenceSubsystemFacts {
f := inferenceSubsystemFacts{
Disabled: p.isInferenceDisabled != nil && p.isInferenceDisabled(),
UsableEngine: hasUsableEngine(p.registry, hw, p.ollamaUsable),
UsableEngine: hasUsableEngine(p.registry, hw, p.ollamaUsable, p.vllmUsable),
}
if p.ollama != nil {
f.Parked = p.ollama.IsParked()
Expand All @@ -2184,7 +2209,7 @@ func (p *agentInferenceProvider) SubsystemState(ctx context.Context) string {
return subsystemState(p.subsystemFacts(ctx, p.profiler.Profile(ctx), st))
}

func hasUsableEngine(reg *infruntime.Registry, hw hardware.Profile, ollamaUsable func() bool) bool {
func hasUsableEngine(reg *infruntime.Registry, hw hardware.Profile, ollamaUsable, vllmUsable func() bool) bool {
for _, name := range reg.Names() {
switch name {
case "ollama":
Expand All @@ -2200,7 +2225,17 @@ func hasUsableEngine(reg *infruntime.Registry, hw hardware.Profile, ollamaUsable
return true
}
case "vllm":
if hw.Engines.VLLM.Installed {
// Same shape, and it did not used to have one (#225): this
// arm read the profile directly, which made it the last
// engine-presence site not asking engineInstalledOnHost. The
// profile resolves the venv now (#238) but is cached for 30 s,
// so it is late for a venv that appeared during setup — the
// freshness engine_resolve.go exists to provide.
if vllmUsable != nil {
if vllmUsable() {
return true
}
} else if hw.Engines.VLLM.Installed {
return true
}
}
Expand Down
70 changes: 57 additions & 13 deletions cmd/waired-agent/inference_backend_probe.go
Original file line number Diff line number Diff line change
Expand Up @@ -56,15 +56,45 @@ type gpuEngagement struct {
// next backend when the current one is CPU-bound. It returns the backend
// the engine ended up on (informational; surfaced by the caller).
//
// Conservative by design (see file header): a single-step plan returns
// immediately without touching the engine, and any inconclusive check
// keeps the current backend.
// EVERY plan is verified, not only the ones with somewhere to fall back
// to (#70). Until then this returned immediately unless plan.Probes(),
// so a detected GPU that failed to ENGAGE — a broken CUDA runtime, VRAM
// already exhausted — kept its GPU label while inference ran on the CPU,
// which is the silent fallback the label exists to make visible. A
// single-step host has no better backend to try, so the correction there
// is the label alone: honest reporting, no restart.
//
// Conservative by design (see file header): any inconclusive check keeps
// the current backend, and only positive evidence of CPU-only residency
// changes anything.
func resolveBackendWithProbe(ctx context.Context, sw backendSwitcher, plan infruntime.BackendPlan, baseURL string, client *http.Client, logger *slog.Logger) infruntime.OllamaBackend {
if !plan.Probes() {
return plan.Preferred().Backend
// A plan with no steps has nothing to report. ResolveOllamaBackend
// always returns at least one, so this is the zero value — a provider
// built without a boot plan. It matters because Preferred() indexes
// Steps[0] unguarded, and the Probes() gate that used to stand at the
// call site was incidentally shielding it: !Probes() is also true for
// an empty plan. "" is what ResolvedBackend already means by "not
// decided", and the caller declines to overwrite a label with it.
if len(plan.Steps) == 0 {
return ""
}
preferred := plan.Preferred().Backend
// A plan that already says CPU has no GPU claim to be wrong about,
// and nothing below it to fall to. Probing it could only cost a
// request and reach the same answer.
if preferred == infruntime.BackendCPU {
return preferred
}
// Loading a model just to read its residency is only worth it when a
// restart could follow, so multi-step plans keep exactly the
// behaviour they had. On a single-step plan the probe is read-only:
// the outcome is a label rather than a restart, a cold load costs up
// to probeLoadTimeout, and the engine can restart under a screen on
// its own — forcing a load into that window would be the "make it
// worse" this file forbids.
mayLoad := plan.Probes()
for i, step := range plan.Steps {
eng := ollamaEngagement(ctx, client, baseURL)
eng := ollamaEngagement(ctx, client, baseURL, mayLoad)
switch {
case !eng.Checked:
logger.Warn("ollama GPU engagement unverified; keeping backend",
Expand All @@ -76,8 +106,13 @@ func resolveBackendWithProbe(ctx context.Context, sw backendSwitcher, plan infru
}
// Positive evidence the model is CPU-resident.
if i == len(plan.Steps)-1 {
logger.Warn("ollama still CPU-bound after exhausting GPU backends; running on CPU",
"backend", step.Backend, "detail", eng.Detail)
if plan.Probes() {
logger.Warn("ollama still CPU-bound after exhausting GPU backends; running on CPU",
"backend", step.Backend, "detail", eng.Detail)
} else {
logger.Warn("ollama did not engage the GPU and has no fallback backend; reporting CPU",
"backend", step.Backend, "detail", eng.Detail)
}
return infruntime.BackendCPU
}
next := plan.Steps[i+1]
Expand All @@ -99,14 +134,23 @@ func resolveBackendWithProbe(ctx context.Context, sw backendSwitcher, plan infru
}

// ollamaEngagement reports whether a model is currently resident on the
// GPU. It inspects /api/ps first; if nothing is loaded it loads the first
// available tag (POST /api/generate with model only) and re-inspects.
// Checked is false when no model could be loaded — the caller must treat
// that as "unknown" and NOT trigger a fallback.
func ollamaEngagement(ctx context.Context, client *http.Client, baseURL string) gpuEngagement {
// GPU. It inspects /api/ps first; if nothing is loaded and mayLoad is
// set it loads the first available tag (POST /api/generate with model
// only) and re-inspects. Checked is false when no model could be read —
// the caller must treat that as "unknown" and NOT trigger a fallback.
//
// mayLoad is false for a plan whose verdict can only relabel, never
// restart (#70). Reading an already-resident model is a cheap request;
// forcing a cold load is minutes, and the boot path has already warmed
// the serving model by the time this runs, so the read usually answers
// on its own.
func ollamaEngagement(ctx context.Context, client *http.Client, baseURL string, mayLoad bool) gpuEngagement {
if eng, ok := psEngagement(ctx, client, baseURL); ok {
return eng
}
if !mayLoad {
return gpuEngagement{Detail: "no model resident to read GPU engagement from"}
}
tag, err := firstOllamaTag(ctx, client, baseURL)
if err != nil || tag == "" {
return gpuEngagement{Detail: "no model available to probe GPU engagement"}
Expand Down
Loading
Loading