From 2d9d301441f6d74dca034c45aa98545918520cac Mon Sep 17 00:00:00 2001 From: Tanguille <91473554+Tanguille@users.noreply.github.com> Date: Mon, 27 Jul 2026 17:49:29 +0200 Subject: [PATCH 1/8] chore(ai): drop restated defaults and dead config, add missing schema headers The ai namespace had accumulated config that restates what the API server or the chart already supplies, plus comments that outlived the thing they described. Removed only fields proven inert: - VirtualMCPServer serviceType/db and HTTPRoute parentRefs group/kind + PathPrefix matches are CRD/Gateway-API defaults. Confirmed with `kubectl apply --dry-run=server`: the API server refills every one, so the stored objects are unchanged. `config.aggregation` was deliberately NOT removed - it is not defaulted when the parent object is absent, and it controls tool-name prefixing. - llmkube-models `timeout: 60m` never applied: the cluster-apps patch strategic-merges `timeout: 5m` onto every child Kustomization. `flate diff` shows no rendered change from deleting it. - llmkube `prometheusRule.enabled: false` and litellm's three secretKeyRef blocks restate a chart default and an envFrom that the CRD supports. - memini dependsOn llmkube-models was already implied through litellm. Comment fixes: the model-cache comment named the two iGPU nodes backwards (qwen3-embedding is on control-2, not control-3), a memory note cited a revision sha that appears nowhere in the repo, and the /dev/dri supplementalGroups block was explained verbatim in three files. Repo-wide consistency: 115 missing `yaml-language-server` schema headers added across 71 files (core/v1 kinds skipped - that host serves no schema for them), and the nine raw-githubusercontent kustomization schema URLs normalised to json.schemastore.org. Docs: deleted the completed sglang-oci-cutover plan per its own exit criterion, repointed three links at a directory that no longer exists, and corrected the sglang pin from v0.5.14 to the v0.5.15 actually running. --- docs/llm-hosting/embedder-cpu-vs-igpu.md | 5 +- docs/llm-hosting/sglang-benchmarks.md | 43 ++--------- docs/llm-hosting/sglang-blockers.md | 8 +- docs/llm-hosting/sglang-oci-cutover.md | 74 ------------------- docs/llm-hosting/vllm-vs-sglang-2026-07.md | 7 +- .../runners/cluster/rbac.yaml | 1 + .../apps/ai/litellm/instance/models.yaml | 7 +- .../apps/ai/litellm/instance/proxy.yaml | 37 +++------- kubernetes/apps/ai/litellm/ks.yaml | 1 - .../apps/ai/llmkube/app/helmrelease.yaml | 31 +++----- kubernetes/apps/ai/llmkube/ks.yaml | 7 +- .../ai/llmkube/models/qwen3-embedding.yaml | 17 ++--- .../apps/ai/llmkube/models/qwen35-2b.yaml | 9 +-- .../ai/llmkube/models/qwen36-27b-sglang.yaml | 2 + .../ai/llmkube/models/qwen36-27b-vllm.yaml | 2 + .../ai/llmkube/models/vmcp-embedding.yaml | 52 +++++-------- kubernetes/apps/ai/memini/ks.yaml | 3 +- .../apps/ai/odysseus/app/helmrelease.yaml | 2 +- .../apps/ai/odysseus/app/kustomization.yaml | 2 +- .../apps/ai/opencode/app/helmrelease.yaml | 3 +- .../ai/opencode/app/opencode-vm-cors.yaml | 1 + .../app/opencode-vm-endpointslice.yaml | 1 + .../opencode/app/opencode-vm-httproute.yaml | 1 + .../apps/ai/toolhive/app/helmrelease.yaml | 1 + .../apps/ai/toolhive/app/httproute.yaml | 70 ++++-------------- .../apps/ai/toolhive/app/ocirepository.yaml | 1 + .../apps/ai/toolhive/config/context7.yaml | 1 + .../ai/toolhive/config/flux-operator.yaml | 3 + .../apps/ai/toolhive/config/github.yaml | 3 + .../apps/ai/toolhive/config/grafana.yaml | 9 ++- .../ai/toolhive/config/homeassistant.yaml | 4 +- .../apps/ai/toolhive/config/karakeep.yaml | 4 +- .../apps/ai/toolhive/config/kubesearch.yaml | 3 + .../apps/ai/toolhive/config/mcpgroups.yaml | 9 ++- .../apps/ai/toolhive/config/postgres-mcp.yaml | 3 + .../apps/ai/toolhive/config/searxng.yaml | 5 +- .../apps/ai/toolhive/config/talos-mcp.yaml | 4 + .../ai/toolhive/config/virtualmcpservers.yaml | 34 +++------ .../apps/ai/toolhive/crds/helmrelease.yaml | 1 + .../apps/ai/toolhive/crds/ocirepository.yaml | 1 + kubernetes/apps/ai/toolhive/ks.yaml | 19 +++-- .../barman-cloud/helmrelease.yaml | 1 + .../cloudnative-pg/cluster/pooler-ro.yaml | 1 + .../cluster/pooler-session.yaml | 1 + .../cloudnative-pg/cluster/pooler.yaml | 1 + .../apps/database/dragonfly/cluster/pdb.yaml | 1 + .../machine-learning/kustomization.yaml | 2 +- .../default/immich/server/kustomization.yaml | 2 +- .../default/nextcloud/app/kustomization.yaml | 2 +- .../facerecognition/kustomization.yaml | 2 +- .../nextcloud/notify-push/kustomization.yaml | 2 +- .../apps/default/obico/app/kustomization.yaml | 2 +- .../default/picoshare/app/kustomization.yaml | 2 +- kubernetes/apps/default/searxng/app/pdb.yaml | 1 + .../default/spoolman/app/kustomization.yaml | 2 +- .../flux-operator/app/clusterrolebinding.yaml | 1 + .../apps/kube-system/amdgpu-undervolt.yaml | 1 + .../apps/kube-system/priorityclass.yaml | 1 + .../apps/network/cloudflared/app/pdb.yaml | 1 + .../external-service/arm/endpointslice.yaml | 1 + .../external-service/arm/httproute.yaml | 1 + .../external-service/avr/endpointslice.yaml | 1 + .../external-service/avr/httproute.yaml | 1 + .../centauri-carbon/endpointslice.yaml | 1 + .../centauri-carbon/httproute.yaml | 1 + .../homeassistant/endpointslice.yaml | 1 + .../homeassistant/httproute.yaml | 1 + .../external-service/ipmi/endpointslice.yaml | 1 + .../external-service/ipmi/tlsroute.yaml | 1 + .../external-service/ntopg/endpointslice.yaml | 1 + .../external-service/ntopg/httproute.yaml | 1 + .../opnsense/endpointslice.yaml | 1 + .../external-service/opnsense/tlsroute.yaml | 1 + .../scrutiny/endpointslice.yaml | 1 + .../external-service/scrutiny/httproute.yaml | 1 + .../truenas/endpointslice.yaml | 1 + .../external-service/truenas/tlsroute.yaml | 1 + .../vaultwarden/endpointslice.yaml | 1 + .../vaultwarden/httproute.yaml | 1 + .../exporters/nut-exporter/ks.yaml | 1 + .../exporters/prowlarr-exporter/ks.yaml | 1 + .../exporters/qbittorrent-exporter/ks.yaml | 1 + .../exporters/sonarr-exporter/ks.yaml | 1 + .../exporters/speedtest-exporter/ks.yaml | 1 + .../observability/kromgo/app/helmrelease.yaml | 1 + .../kromgo/app/ocirepository.yaml | 1 + .../observability/siren/app/helmrelease.yaml | 1 + kubernetes/apps/observability/siren/ks.yaml | 1 + .../bouncers/envoy/referencegrant.yaml | 1 + .../apps/web3/monero/xmrig/priorityclass.yaml | 1 + 90 files changed, 212 insertions(+), 336 deletions(-) delete mode 100644 docs/llm-hosting/sglang-oci-cutover.md diff --git a/docs/llm-hosting/embedder-cpu-vs-igpu.md b/docs/llm-hosting/embedder-cpu-vs-igpu.md index bddb964725..fdb5511d84 100644 --- a/docs/llm-hosting/embedder-cpu-vs-igpu.md +++ b/docs/llm-hosting/embedder-cpu-vs-igpu.md @@ -51,9 +51,8 @@ stays, along with the nodeSelectors / affinity rules that only make sense while the embedders are GPU-pinned. The leak is bounded by LLMKube's native `maxPodLifetimeSeconds: 86400` on both -InferenceServices. This support landed in upstream PR #1182 and is available -in chart/CRD version 0.9.10; the chart and CRDs must be upgraded before these -model fields are reconciled. LLMKube copies the lifetime to the generated +InferenceServices. This support landed in upstream PR #1182 and has been +live since chart/CRD 0.9.10. LLMKube copies the lifetime to the generated Deployment pod template, so each Vulkan embedder is recycled after 24 hours, including startup and model load time, without bespoke controller logic. node-exporter's drm collector plus alerts on pinned GTT >4 GiB, MemAvailable diff --git a/docs/llm-hosting/sglang-benchmarks.md b/docs/llm-hosting/sglang-benchmarks.md index aed73f7fa6..c98eb01ce9 100644 --- a/docs/llm-hosting/sglang-benchmarks.md +++ b/docs/llm-hosting/sglang-benchmarks.md @@ -1,12 +1,12 @@ # Qwen3.6-27B serving on R9700 (gfx1201) — Benchmarks & Decision **SUPERSEDED:** production runs SGLang (baked `ghcr.io/tanguille/sglang-rdna4` image) — see -`docs/llm-hosting/sglang-blockers.md` and `kubernetes/apps/ai/sglang/README.md` for the current +`docs/llm-hosting/sglang-blockers.md` and `docker/sglang-rdna4/README.md` for the current engine decision. This doc is the historical vLLM-vs-SGLang record from 2026-06-21. ## Round 2 Final Results — 2026-06-21 -**All experiments complete. Best engine: vLLM kyuz0 + cyankiwi AWQ-INT4 + MTP×4 + 234K context.** +**As measured on 2026-06-21 (since superseded): vLLM kyuz0 + cyankiwi AWQ-INT4 + MTP×4 + 234K context.** ### SGLang 0.5.13 — CANNOT serve Qwen3.6-27B on gfx1201 @@ -54,17 +54,17 @@ reference. vLLM 0.23.0 cannot match it for this model due to the breaking V1 eng --- -## FINAL DECISION — vLLM kyuz0 + MTP×4 + 234K context (2026-06-21) +## Decision as of 2026-06-21 (superseded — production is SGLang) -Updated from prior llama.cpp recommendation. vLLM kyuz0 with speculative MTP decoding is the production -engine for multi-user workloads. llama.cpp retained only for single-user interactive work +Updated from prior llama.cpp recommendation. vLLM kyuz0 with speculative MTP decoding was the chosen +engine for multi-user workloads at the time. llama.cpp retained only for single-user interactive work (better C=1 TG: 39 vs 19.5 tok/s). ### Complete comparison table (all confirmed on gfx1201) | Engine | TG C=1 | PP tok/s | Agg TG C=8 | Max ctx | APC | MTP | Note | |---|---:|---:|---:|---:|:---:|:---:|---| -| **vLLM kyuz0 + MTP×4, 234K** | **19.5** | ~1200 | **126.1** | **234,320** | ✅ | ✅ | **PRODUCTION** (multi-user) | +| **vLLM kyuz0 + MTP×4, 234K** | **19.5** | ~1200 | **126.1** | **234,320** | ✅ | ✅ | Chosen 2026-06-21 (multi-user) | | vLLM kyuz0 + MTP×4, 131K | 18.3 | ~1200 | 129.6 | 131,072 | ✅ | ✅ | Better C=4 throughput (70.4 vs 61.2) | | vLLM kyuz0 + MTP×4, 32K | 18.4 | ~1200 | 126.8 | 32,768 | ✅ | ✅ | First MTP result (baseline for tuning) | | **llama.cpp ROCm + MTP** | **~39** | ~423 cold | ~35 (flat) | ~256K | ✅ | ✅ | Best single-stream; flat agg at all C | @@ -93,11 +93,7 @@ Why not native 262K context? Gap of 0.93 GiB: 9.48 GiB needed, 8.55 GiB availabl - `--kv-offloading-size native`: doesn't extend GPU KV pool (for weight offload, not KV pool) - `--swap-space`: flag doesn't exist in kyuz0; `--kv-offloading-size` fills this role for weights -Detailed experiment ledger: [`vllm-optimization-ledger.md`](./vllm-optimization-ledger.md) - -Retired trial configs (revival anchors) are in -[`ai-serving-trials-archive.md`](./ai-serving-trials-archive.md). The rest of this file is the -**historical sglang/vLLM measurement log** that led to the decision. +The rest of this file is the **historical sglang/vLLM measurement log** that led to the decision. --- @@ -497,28 +493,3 @@ verify). The max concurrency sweet spot for QoS is C=8. **Note:** vLLM auto-set `max_num_scheduled_tokens=2048` based on spec config. This limits token budget per scheduling step. No impact observed at C≤16 with out=200. - ---- - -## Open / next experiments - -- [x] **bf16 KV cache** — TESTED: *slower* (12.35 vs 14.96) and half capacity. fp8 KV wins on - sglang (native gfx1201 fp8 kernels). Keep fp8_e4m3. -- [x] vLLM kyuz0 batching benchmark — **DONE**: 48.6 tok/s @ C8, 85.8 tok/s @ C16, APC on, CUDAGraphs work on gfx1201 in 0.22.1rc1 (see section above) -- [x] vLLM FP8 KV + APC coexistence — **DONE**: 128.9 tok/s @ C32 (50% gain over FP16 baseline); bug #13147 doesn't affect kyuz0. FP8 is recommended config. -- [x] VLLM_ROCM_USE_AITER=1 — **FAILS**: GDN kernel autotune exhausts all configs ≤ 64 KiB SRAM on gfx1201. AITER config tables target MI300X, not gfx1201. -- [x] **MTP speculative decoding** — **DONE**: `--spec-method mtp --spec-tokens 4`. C=1: 18.4 tok/s (2.6×), C=8: 126.8 tok/s agg (2.7×). Qwen3.6-27B has built-in MTP heads — no separate draft model needed. MTP is now the recommended production config. -- [ ] vLLM fp8 KV + extended context: `--max-model-len 131072` with fp8 KV — test if 32K→131K context affects throughput -- [ ] Push MTP C beyond 16 with max_num_seqs=32 — test if C=32 MTP reaches 200+ tok/s -- [ ] `mamba_track_interval` / scheduler tuning under Config B for higher concurrency. -- [ ] Quality probe (logprob/eval) to fully close the fork's flagged torch-2.12 attention drift - (basic coherence already passes). -- [ ] **llama.cpp gfx1201 tuning sweep** — `-b/-ub`, hipBLASLt env, perf-level, iGPU/`-sm`, - Vulkan backend (see "llama.cpp on gfx1201" section). Then cache probe + concurrency sweep. - -## Process Instructions - -- After completing each step, update the relevant table with the current status. -- Pause for user confirmation before proceeding to next step. -- Keep this as a living reference; record the winning config + rationale once chosen and - committed to the HelmRelease. diff --git a/docs/llm-hosting/sglang-blockers.md b/docs/llm-hosting/sglang-blockers.md index 1469796fbe..286c609dc1 100644 --- a/docs/llm-hosting/sglang-blockers.md +++ b/docs/llm-hosting/sglang-blockers.md @@ -2,7 +2,7 @@ Tracking what needs to land upstream before SGLang can replace vLLM in production without depending on the mattbucci RDNA4 fork. -**Current approach:** `mattbucci/2x-R9700-RDNA4-GFX1201-sglang-inference` fork, **v0.5.14** (torch 2.11+rocm7.2, Triton 3.6), baked into the `ghcr.io/tanguille/sglang-rdna4` image by `.github/workflows/build-sglang-rdna4.yaml`. Build/pin/rollback mechanics are in `docker/sglang-rdna4/README.md`. `kubernetes/apps/ai/sglang/app/scripts/sglang-env-rebuild.sh` is the retired PVC-rebuild recipe, kept only as an emergency fallback. +**Current approach:** `mattbucci/2x-R9700-RDNA4-GFX1201-sglang-inference` fork, **v0.5.15** (torch 2.11+rocm7.2, Triton 3.6), baked into the `ghcr.io/tanguille/sglang-rdna4` image by `.github/workflows/build-sglang-rdna4.yaml`. Build/pin/rollback mechanics are in `docker/sglang-rdna4/README.md`. The retired PVC-rebuild recipe is gone with the `sglang` app directory; recover it from git history if ever needed. --- @@ -147,7 +147,7 @@ mitigation is obsolete. **Required RDNA4/TP=1 patch (NOT in the fork — it targets a dual-card TP=2 box):** the JIT `store_cache` kernel aborts at TP=1 (`kvcache.cuh:204: CUDA error: the operation cannot be performed in the present state`). Force `can_use_store_cache()->False` (naive torch KV store). A stock -`setup.sh` rebuild without this crashes on the first request — captured in `kubernetes/apps/ai/sglang/app/scripts/sglang-env-rebuild.sh`. +`setup.sh` rebuild without this crashes on the first request. A **second RDNA4/TP=1 patch** lives in the same recipe: the sampler's cross-TP token-id all-reduce (`_sync_token_ids_across_tp`) runs even at TP=1 — a no-op on a 1-rank group — for grammar/structured-output @@ -253,7 +253,7 @@ not the kernel, that makes overlap a no-op). **Root cause:** 067's hunk expects `triton_attention_split_tile_size: Optional[int] = None` to be immediately followed by `num_continuous_decode_steps: int = 1` in the `ServerArgs` dataclass, so it can insert `force_decode_window` between them. Our pinned tree (`FORK_REF=60ffa9501c2c6`) already has four newer fields inserted at that exact location (`prefill_only_disable_kv_cache`, `disable_radix_cache`, `disable_chunked_prefix_cache`, `disable_overlap_schedule`) from later upstream/fork changes the `.CANDIDATE` series was never rebased against. This is real semantic drift, not a cosmetic line-number offset — `patch`'s fuzz matching (already at fuzz=2) can't resolve it. -**Investigated:** 2026-07-01, caught at the dry-run phase with zero production impact — full detail in [[project_sglang_decode_topk_patch069]] and `docs/llm-hosting/decode-topk-pages-test-plan.md` (architectural compatibility with the DeltaNet hybrid was confirmed separately; the patches are sound, just stale against our tree). +**Investigated:** 2026-07-01, caught at the dry-run phase with zero production impact — full detail in the blocker 7 row below (architectural compatibility with the DeltaNet hybrid was confirmed separately; the patches are sound, just stale against our tree). **Status:** NO-GO per the test plan's gate — no hand-rebasing the patches (upstream hasn't validated a hand-patched variant). `.CANDIDATE` files and `FORK_REF` left untouched. Retest when upstream rebases the `.CANDIDATE` series past our pin, or incidentally on our next `FORK_REF` bump. @@ -266,7 +266,7 @@ not the kernel, that makes overlap a no-op). **Context:** AMD Quark 0.12.0 (2026-07) adds AMDFP4 (E5M3 per-block scales), NVFP4, native MXFP4 inference support, and the SVDQuant algorithm (INT4/MXFP4/NVFP4 weights + low-rank outlier branch). Tempting as a path off the slow AWQ-int4 dense-decode wall. **Root cause (why none of it helps this box):** Quark is a *quantization producer* — it emits checkpoints. Our dense-decode bottleneck is the RDNA4 **inference kernel** (Triton W4A16 + GatedDeltaNet), an SGLang/fork problem the checkpoint format can't touch. Re-quantizing the same weights to a different FP4 format doesn't change which kernel SGLang dispatches at decode. Producer with no consumer: -- **MXFP4 / NVFP4 / AMDFP4:** no fused decode kernel in the mattbucci fork for gfx1201 (consistent with the MXFP4/fp4 "ruled out for dense" finding in `sglang-benchmarks.md` and [[project_sglang_rdna4_tp1_results]]). +- **MXFP4 / NVFP4 / AMDFP4:** no fused decode kernel in the mattbucci fork for gfx1201 (consistent with the MXFP4/fp4 "ruled out for dense" finding in `sglang-benchmarks.md`). - **SVDQuant:** its runtime is **Nunchaku — CUDA W4A4 + low-rank, Nvidia-only**; no ROCm path, and SGLang has no consumer for the SVDQuant (INT4 + low-rank branch) format on any GPU. Public SVDQuant checkpoints are also almost all diffusion/image models (FLUX, Qwen-Image); no LLM checkpoint exists for Qwen3.6-27B (even Qwen3.5-27B is GPTQ-Int4/AWQ only). **Investigated:** 2026-07-04, doc-only (no build). SVDQuant quality-vs-AWQ was the one non-throughput angle (better int4 accuracy at equal speed) but we've flagged no quality issue, so not worth the self-quantize + missing-kernel effort. diff --git a/docs/llm-hosting/sglang-oci-cutover.md b/docs/llm-hosting/sglang-oci-cutover.md deleted file mode 100644 index 94a9b31574..0000000000 --- a/docs/llm-hosting/sglang-oci-cutover.md +++ /dev/null @@ -1,74 +0,0 @@ -# SGLang: runtime-from-PVC → baked OCI image - -Move sglang from the **runtime-from-PVC workaround** (engine built out-of-band onto the -`sglang` PVC by `scripts/sglang-env-rebuild.sh`, image = bare ROCm base) to a **git-reproducible -baked image** `ghcr.io/tanguille/sglang-rdna4`. The PVC keeps `/cache/hf` and -`/cache/sglang/triton`; the old runtime directories are retired. - -**Why now:** the workaround existed because the cluster couldn't pull never-cached images -(Spegel upstream fall-through was broken). Spegel is healthy again, so the README's documented -follow-up — bake the env into a versioned image — is unblocked. The win is reproducibility -(rebuild-from-Dockerfile, immutable digest, self-healing Deployment) instead of a hand-built PVC. - -## State of play - -- The previously-pushed image is stale **`:v0.5.12-gfx1201`** — rebuilt at v0.5.14 by Stage 1. -- The build infra (Dockerfile, workflow) was prototyped in a worktree branch and is brought - current + onto `main` by Stage 1. -- `docker/sglang-rdna4/Dockerfile` mirrors `sglang-env-rebuild.sh` (same `FORK_REF`, - `SGLANG_TAG`, the 3 TP=1 patches). The script stays as the emergency PVC-rebuild fallback. - -## Stage 1 — build infra (this PR, no production impact) - -- `docker/sglang-rdna4/` — Dockerfile (v0.5.14 + 3 TP=1 patches), entrypoint, README. -- `.github/workflows/build-sglang-rdna4.yaml` — builds + pushes `v0.5.14-gfx1201` on - **`ubuntu-latest`** (GPU-free — rationale in `docker/sglang-rdna4/README.md`). - -Building off-cluster means **no maintenance window, ever**: builds never touch control-1's -RAM/VRAM or live serving. The trade-off: a broken kernel build surfaces at pod boot instead of -at image-build time — covered by digest-pinned rollback and the Stage 2 validation below. - -## Stage 2 — the cutover (no maintenance window) - -1. **Build** — auto-fires on the Stage 1 merge, or `gh workflow run build-sglang-rdna4.yaml`. - Watch `gh run watch`. ~15-30 min; serving keeps running throughout. Capture the pushed - digest from the run summary (`…@sha256:…`). First run may hit hosted-runner disk/RAM - limits — iterate on the workflow's free-disk-space/swap knobs if so. -2. **Pin** — set the sglang HelmRelease image to `ghcr.io/tanguille/sglang-rdna4` with - `tag: v0.5.14-gfx1201@sha256:` (tag@digest, repo convention: the digest is - authoritative — the tag gets rebuilt in place — but the tag must stay so Renovate can bump - the digest), command path `/cache/sglang/repo-v0514/scripts/launch.sh` → - `/opt/rdna4-inference/scripts/launch.sh`, and drop the `CONDA_BASE` env override (the image - bakes `/opt/conda`). Keep the PVC mount (model `/cache/hf` + Triton `/cache/sglang/triton`) - and every other flag identical. Open as Stage 2 PR. -3. **Pre-pull** — control-1 is the only node that can run the pod, so Spegel has no peer for - the first pull and Recreate kills the old pod before pulling: an un-prepulled 47GB fetch is - pure downtime. Before merging: `kubectl -n ai run prepull -i --rm --restart=Never - --image=ghcr.io/tanguille/sglang-rdna4:v0.5.14-gfx1201@sha256: - --overrides='{"spec":{"nodeName":"control-1"}}' --command -- true` (or `crictl pull` on the - node). Repeat for every future digest bump. -4. **Deploy** — merge Stage 2, watch the pod roll (this restart is the only serving-visible - moment of the whole cutover). First boot recompiles the Triton JIT cache onto the PVC (the - startup probe's ~16 min budget covers it). This boot is also the kernel smoke test the - GPU-free build deferred: a broken build fails the startup probe here. The single GPU forces - a Recreate rollout (no old pod kept warm), so on failure go straight to **Rollback** below — - serving is down only for the failed-boot window either way. -5. **Validate** — `/health` up, a `qwen-3.6` completion with tools+thinking, PPL/needle spot-check, decode tok/s parity with the PVC baseline (≈16 single / ≈99 @conc24). -6. **Retire** — the PVC's `/cache/sglang/conda` and `/cache/sglang/repo-v0514` are retired and may - be deleted separately. Preserve `/cache/hf` and `/cache/sglang/triton`. Keep - `app/scripts/sglang-env-rebuild.sh` as the emergency PVC-environment rebuild path. - -**Rollback:** after the retired runtime directories are deleted, rollback requires the baked OCI -image or rebuilding the emergency PVC environment with `app/scripts/sglang-env-rebuild.sh`. - -## Process Instructions - -- After completing each step, update the plan with the current status. -- Pause for user confirmation before proceeding to next step. -- Suggest the prompt for continuing to the next step. -- After the last step, make a final documentation pass. Once the contents of the plan have been - consolidated into existing documentation (the app README + `docker/sglang-rdna4/README.md`), - this plan file can be removed. - -**Important:** Every prompt should verify the branch and worktree before doing any work -(`main` was 16 commits behind during this work — always `git pull && git status` first). diff --git a/docs/llm-hosting/vllm-vs-sglang-2026-07.md b/docs/llm-hosting/vllm-vs-sglang-2026-07.md index 18c1d507d2..b9d9bbfc89 100644 --- a/docs/llm-hosting/vllm-vs-sglang-2026-07.md +++ b/docs/llm-hosting/vllm-vs-sglang-2026-07.md @@ -204,11 +204,8 @@ file before deciding to transfer. Scripts live in `bench/` next to this doc, so the numbers above stay auditable: -- `concsweep.py` — aggregate decode at concurrency 1/8/16. Short prompts to isolate - decode from prefill, unique salt per stream so prefix caching cannot inflate results. -- `spectest.py` — verbatim-reproduction throughput on a 13.8K-token prompt. This is the - coding-agent shape where prompt-lookup n-gram actually hits; a short-prompt test shows - near-zero acceptance. +- `concsweep.py` — aggregate decode at concurrency 1/8/16. +- `spectest.py` — verbatim-reproduction throughput on a 13.8K-token prompt. Both drive the OpenAI `/v1/completions` endpoint, so they run unmodified against either engine. Port-forward the service and pass the port: diff --git a/kubernetes/apps/actions-runner-system/actions-runner-controller/runners/cluster/rbac.yaml b/kubernetes/apps/actions-runner-system/actions-runner-controller/runners/cluster/rbac.yaml index 1b6ee482b7..ad0a6c8249 100644 --- a/kubernetes/apps/actions-runner-system/actions-runner-controller/runners/cluster/rbac.yaml +++ b/kubernetes/apps/actions-runner-system/actions-runner-controller/runners/cluster/rbac.yaml @@ -4,6 +4,7 @@ kind: ServiceAccount metadata: name: cluster-runner --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/talos.dev/serviceaccount_v1alpha1.json apiVersion: talos.dev/v1alpha1 kind: ServiceAccount metadata: diff --git a/kubernetes/apps/ai/litellm/instance/models.yaml b/kubernetes/apps/ai/litellm/instance/models.yaml index c98aca6b2a..a8be77b6e8 100644 --- a/kubernetes/apps/ai/litellm/instance/models.yaml +++ b/kubernetes/apps/ai/litellm/instance/models.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/litellm.home-operations.com/litellmmodel_v1alpha1.json apiVersion: litellm.home-operations.com/v1alpha1 kind: LiteLLMModel metadata: @@ -22,6 +23,7 @@ spec: maxInputTokens: 171808 maxOutputTokens: 8192 --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/litellm.home-operations.com/litellmmodel_v1alpha1.json apiVersion: litellm.home-operations.com/v1alpha1 kind: LiteLLMModel metadata: @@ -34,8 +36,8 @@ spec: apiBase: http://qwen36-27b.ai.svc.cluster.local:30000/v1 apiKey: sk-sglang-noauth additional: - # 900s (was 600): Hermes compression is a ~100K+ prefill on this alias — under GPU - # contention the shared prefill rate drops below what 600s covers (observed 2026-07-07). + # Hermes compression is a ~100K+ prefill on this alias; under GPU contention + # the shared prefill rate drops below what 600s covers (2026-07-07). timeout: 900 stream_timeout: 900 num_retries: 0 @@ -49,6 +51,5 @@ spec: enable_thinking: false info: mode: chat - # Match SGLang's effective 180K context cap after reserving maxOutputTokens. maxInputTokens: 171808 maxOutputTokens: 8192 diff --git a/kubernetes/apps/ai/litellm/instance/proxy.yaml b/kubernetes/apps/ai/litellm/instance/proxy.yaml index 4d1032aacd..25327c4d79 100644 --- a/kubernetes/apps/ai/litellm/instance/proxy.yaml +++ b/kubernetes/apps/ai/litellm/instance/proxy.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/litellm.home-operations.com/litellmproxy_v1alpha1.json apiVersion: litellm.home-operations.com/v1alpha1 kind: LiteLLMProxy metadata: @@ -13,25 +14,10 @@ spec: key: LITELLM_MASTER_KEY service: port: 80 - env: - - name: LITELLM_MASTER_KEY - valueFrom: - secretKeyRef: - name: litellm-secret - key: LITELLM_MASTER_KEY - - name: LITELLM_SALT_KEY - valueFrom: - secretKeyRef: - name: litellm-secret - key: LITELLM_SALT_KEY - - name: DATABASE_URL - valueFrom: - secretKeyRef: - name: litellm-secret - key: DATABASE_URL - # Chart defaults (10s initialDelay, 3x failureThreshold ~= 40s grace) kill the - # container mid prisma-migrate on first boot. Measured cold start: ~81s - # (pod creation to "Application startup complete"); this budgets ~120s. + envFrom: + - secretRef: + name: litellm-secret + # Measured cold start ~81s; the chart's ~40s grace kills prisma-migrate. livenessProbe: httpGet: path: /health/liveliness @@ -51,9 +37,8 @@ spec: litellmSettings: callbacks: ["prometheus"] service_callback: ["prometheus_system"] - # Cache embeddings only (chat is near-unique per turn, ~0 hit rate). vmcp re-embeds the - # same ~194 tool descriptions every MCP initialize; caching turns that into dragonfly hits - # that survive embedder restarts, fixing the init timeouts. Auth-less, shared with toolhive. + # Embeddings only — chat is near-unique per turn (~0 hit rate). vmcp re-embeds + # the same ~194 tool descriptions every MCP initialize. cache: true cache_params: type: redis @@ -64,11 +49,9 @@ spec: drop_params: true num_retries: 0 routerSettings: - # Per-error override of the global num_retries: 0. A dropped connection to - # SGLang surfaces as InternalServerError ("OpenAIException - Connection - # error") and costs a whole scheduled run, because Hermes cron has no retry - # queue. Timeouts stay at 0 retries — re-running a ~100K prefill that - # already burned 900s would only deepen the GPU contention that caused it. + # Per-error override of num_retries: 0 — a dropped SGLang connection costs a + # whole Hermes cron run. Timeouts stay at 0: re-running a ~100K prefill that + # already burned 900s only deepens the contention that caused it. retry_policy: InternalServerErrorRetries: 2 route: diff --git a/kubernetes/apps/ai/litellm/ks.yaml b/kubernetes/apps/ai/litellm/ks.yaml index fa90e63182..46072487dd 100644 --- a/kubernetes/apps/ai/litellm/ks.yaml +++ b/kubernetes/apps/ai/litellm/ks.yaml @@ -14,7 +14,6 @@ spec: # Embedding cache backend - name: dragonfly-cluster namespace: database - # Delay the endpoint switch until native SGLang is Available. - name: llmkube-models namespace: ai path: ./kubernetes/apps/ai/litellm/instance diff --git a/kubernetes/apps/ai/llmkube/app/helmrelease.yaml b/kubernetes/apps/ai/llmkube/app/helmrelease.yaml index c448a79e47..8d25878793 100644 --- a/kubernetes/apps/ai/llmkube/app/helmrelease.yaml +++ b/kubernetes/apps/ai/llmkube/app/helmrelease.yaml @@ -22,12 +22,11 @@ spec: limits: memory: 128Mi - # Observability (auto-converted to VM*Scrapes by the cluster's converter): - # - inferencePodMonitor scrapes llama-server /metrics on the inference pods - # (the operator injects --metrics into every InferenceService itself) - # - serviceMonitor scrapes the operator; secure:false serves it over plain - # HTTP so VM can scrape without a ServiceAccount token (metrics.enabled - # defaults true). + # Scrapes are auto-converted to VM*Scrapes by the cluster's converter. + # secure:false serves operator metrics over plain HTTP, so VM scrapes without + # a ServiceAccount token. Chart prometheusRule stays off (its default): the + # alerts key on job="llmkube-inference", this cluster scrapes it as + # "ai/llmkube-inference". Repo alerts live in prometheusrule.yaml. metrics: secure: false prometheus: @@ -35,21 +34,13 @@ spec: enabled: true inferencePodMonitor: enabled: true - prometheusRule: - # Chart alerts key on job="llmkube-inference"/"llmkube-controller" but this - # cluster's scrape job is "ai/llmkube-inference", so they can never fire; - # its recording rules have no consumers. Repo alerts: prometheusrule.yaml. - enabled: false - # Persistent model cache on ceph-filesystem (CephFS, RWX). A SINGLE shared - # PVC mounted by the controller-manager and the iGPU InferenceServices, which - # land on different nodes (qwen3-embedding on control-3, qwen35-2b on - # control-2) — so it must be RWX: an RWO volume multi-attach-errors across - # nodes. qwen36-27b brings its own node-local modelCache.claimName instead. - # The cache is cold (write-once on download, read-once at model load, then - # idle — weights live in VRAM), so CephFS latency never touches inference. - # The operator's model-cache-prep init chowns /models as root, so the chown - # succeeds on CephFS (fsGroupPolicy: None) too. + # Shared RWX cache on CephFS: the controller and the iGPU InferenceServices + # land on different nodes, so an RWO volume multi-attach-errors. Access is + # write-once on download then read-once at load, so CephFS latency never + # touches inference. qwen36-27b brings its own node-local modelCache. + # The model-cache-prep init chowns /models as root, so CephFS (fsGroupPolicy: + # None) works too. modelCache: # qwen3-embedding 0.64Gi + qwen35-2b 1.18Gi + vmcp-embedding 0.29Gi ≈ 2.1Gi # + headroom (chart default is 100Gi) diff --git a/kubernetes/apps/ai/llmkube/ks.yaml b/kubernetes/apps/ai/llmkube/ks.yaml index 04aef41719..e7de9bed08 100644 --- a/kubernetes/apps/ai/llmkube/ks.yaml +++ b/kubernetes/apps/ai/llmkube/ks.yaml @@ -14,10 +14,8 @@ spec: name: flux-system namespace: flux-system targetNamespace: *namespace - # Gate the models (llmkube-models) on the operator being installed: the - # HelmRelease only reports Ready once its CRDs are applied and the controller is - # up. Targeted healthCheck rather than spec.wait so readiness keys on just the - # CRD-providing HelmRelease, not a blanket wait across all rendered resources. + # Targeted healthCheck rather than spec.wait: llmkube-models only needs the + # CRD-providing HelmRelease Ready, not every rendered resource. healthChecks: - apiVersion: helm.toolkit.fluxcd.io/v2 kind: HelmRelease @@ -44,7 +42,6 @@ spec: kind: InferenceService name: qwen36-27b namespace: ai - timeout: 60m healthCheckExprs: - apiVersion: inference.llmkube.dev/v1alpha1 kind: InferenceService diff --git a/kubernetes/apps/ai/llmkube/models/qwen3-embedding.yaml b/kubernetes/apps/ai/llmkube/models/qwen3-embedding.yaml index 0a1f0b56d8..f90b29d308 100644 --- a/kubernetes/apps/ai/llmkube/models/qwen3-embedding.yaml +++ b/kubernetes/apps/ai/llmkube/models/qwen3-embedding.yaml @@ -1,16 +1,13 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/inference.llmkube.dev/model_v1alpha1.json apiVersion: inference.llmkube.dev/v1alpha1 kind: Model metadata: name: qwen3-embedding spec: - # Q8_0 over F16: measured 194-item ToolHive batch at ~10s/request isolated, - # 19-52s under 3x concurrent (vmcp + memini both hit this server) — GPU - # compute-bound on the weak iGPU, not slots/CPU (confirmed via kubectl top: - # CPU peaked at 424m of an uncapped budget during the same test). Q8_0 is - # ~half the compute of F16 for negligible quality loss on embedding-only - # (non-generative) similarity search, and embeddings aren't cached/persisted - # across sessions so there's no stale-vector consistency concern either way. + # Q8_0 over F16: this iGPU is compute-bound, and Q8_0 halves the compute for + # negligible loss on similarity search (194-item batch, ~10s isolated / + # 19-52s under 3x concurrent). # Revision-pinned (immutable bytes) — Renovate bumps the sha via the git-refs # manager in .renovaterc.json5; refreshPolicy makes the bump actually re-fetch. source: https://huggingface.co/Qwen/Qwen3-Embedding-0.6B-GGUF/resolve/370f27d7550e0def9b39c1f16d3fbaa13aa67728/Qwen3-Embedding-0.6B-Q8_0.gguf @@ -28,6 +25,7 @@ spec: # nodeSelector below pins this to the iGPU. resourceName: squat.ai/dri --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/inference.llmkube.dev/inferenceservice_v1alpha1.json apiVersion: inference.llmkube.dev/v1alpha1 kind: InferenceService metadata: @@ -70,6 +68,7 @@ spec: topologyKey: kubernetes.io/hostname # Render/video host groups for /dev/dri — the documented Vulkan-runtime path. + # Mirrored by qwen35-2b and vmcp-embedding. podSecurityContext: runAsUser: 0 runAsGroup: 0 @@ -79,8 +78,8 @@ spec: resources: cpu: "500m" # inference is on the iGPU; CPU only tokenises/schedules - # Current 84746bcc6 revision peaked at ~3.1 GiB working set over 7 days; - # reserve 4 GiB so the scheduler accounts for the model's UMA usage. + # ~3.1 GiB peak working set over 7 days; 4 GiB so the scheduler accounts for + # the model's UMA usage. memory: 4Gi endpoint: path: /v1/embeddings diff --git a/kubernetes/apps/ai/llmkube/models/qwen35-2b.yaml b/kubernetes/apps/ai/llmkube/models/qwen35-2b.yaml index 9471893bc0..4cf337bc57 100644 --- a/kubernetes/apps/ai/llmkube/models/qwen35-2b.yaml +++ b/kubernetes/apps/ai/llmkube/models/qwen35-2b.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/inference.llmkube.dev/model_v1alpha1.json apiVersion: inference.llmkube.dev/v1alpha1 kind: Model metadata: @@ -23,6 +24,7 @@ spec: layers: -1 resourceName: squat.ai/dri --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/inference.llmkube.dev/inferenceservice_v1alpha1.json apiVersion: inference.llmkube.dev/v1alpha1 kind: InferenceService metadata: @@ -31,7 +33,6 @@ spec: modelRef: qwen35-2b # Must be recent enough for Qwen3.5's GDN (Gated Delta Networks) architecture; # verified to load `qwen35` and run GDN fully on Vulkan (no CPU fallback). - # Tracked by the llmkube-models Renovate customManager (.renovaterc.json5). image: ghcr.io/ggml-org/llama.cpp:server-vulkan@sha256:eddb40b8a348a652c8829ef036df9bbbae537755ae7026bf6d4b403840cd468e parallelSlots: 2 # Vulkan warmup hangs at parallel=1 (llama.cpp #24307) contextSize: 16384 # GDN keeps KV tiny, so long ctx is cheap; 16K is plenty for offload @@ -59,7 +60,6 @@ spec: app: qwen3-embedding topologyKey: kubernetes.io/hostname - # Render/video host groups for /dev/dri — the documented Vulkan-runtime path. podSecurityContext: runAsUser: 0 runAsGroup: 0 @@ -67,9 +67,8 @@ spec: - 44 - 226 - # No probeOverrides: the operator's llamacpp defaults already probe /health - # with a more forgiving startup window. GPU allocation comes from - # Model.hardware.gpu.count (resources.gpu/gpuMemory are dead fields). + # GPU allocation comes from Model.hardware.gpu.count; resources.gpu/gpuMemory + # are dead fields here. resources: cpu: "500m" # inference is on the iGPU; CPU only tokenises/schedules/samples # ~2.4 GiB resident under load, 405Mi idle working set; requested low diff --git a/kubernetes/apps/ai/llmkube/models/qwen36-27b-sglang.yaml b/kubernetes/apps/ai/llmkube/models/qwen36-27b-sglang.yaml index dec20498f3..a4bfe1d5a5 100644 --- a/kubernetes/apps/ai/llmkube/models/qwen36-27b-sglang.yaml +++ b/kubernetes/apps/ai/llmkube/models/qwen36-27b-sglang.yaml @@ -25,6 +25,7 @@ spec: requests: storage: 10Gi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/inference.llmkube.dev/model_v1alpha1.json apiVersion: inference.llmkube.dev/v1alpha1 kind: Model metadata: @@ -52,6 +53,7 @@ spec: runtime: rocm resourceName: squat.ai/dri --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/inference.llmkube.dev/inferenceservice_v1alpha1.json apiVersion: inference.llmkube.dev/v1alpha1 kind: InferenceService metadata: diff --git a/kubernetes/apps/ai/llmkube/models/qwen36-27b-vllm.yaml b/kubernetes/apps/ai/llmkube/models/qwen36-27b-vllm.yaml index 835f49c0ae..f6dcc47709 100644 --- a/kubernetes/apps/ai/llmkube/models/qwen36-27b-vllm.yaml +++ b/kubernetes/apps/ai/llmkube/models/qwen36-27b-vllm.yaml @@ -26,6 +26,7 @@ spec: requests: storage: 10Gi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/inference.llmkube.dev/model_v1alpha1.json apiVersion: inference.llmkube.dev/v1alpha1 kind: Model metadata: @@ -68,6 +69,7 @@ spec: resourceName: squat.ai/dri memory: 32Gi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/inference.llmkube.dev/inferenceservice_v1alpha1.json apiVersion: inference.llmkube.dev/v1alpha1 kind: InferenceService metadata: diff --git a/kubernetes/apps/ai/llmkube/models/vmcp-embedding.yaml b/kubernetes/apps/ai/llmkube/models/vmcp-embedding.yaml index 428cceb246..100c4f7b48 100644 --- a/kubernetes/apps/ai/llmkube/models/vmcp-embedding.yaml +++ b/kubernetes/apps/ai/llmkube/models/vmcp-embedding.yaml @@ -1,26 +1,17 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/inference.llmkube.dev/model_v1alpha1.json apiVersion: inference.llmkube.dev/v1alpha1 kind: Model metadata: name: vmcp-embedding spec: - # harrier-oss-v1-270m (Microsoft, 2026-03-30): Gemma3-270m-based decoder embedder, - # MIT, 640-dim, 32K native context, MTEB v2 66.5 — dedicated to vmcp-unified's - # tool-description search only, decoupling its per-session ~194-tool/~1166-token - # batch from qwen3-embedding's contention with memini. - # Known tradeoff: harrier is instruction-tuned (query-side prefixes); ToolHive - # sends bare queries, so we run slightly below its benchmark quality — accepted, - # still above the non-instructed alternatives for a 194-doc search space. - # Placement: shares ONE iGPU with qwen3-embedding (both tiny: 292MB + 639MB Q8; - # the iGPU generic-device-plugin advertises count: 2 for exactly this pair). - # Dedicated server = own slots/queue, so vmcp's session-build batch no longer - # queues behind memini/find_tool traffic at the model level; raw iGPU compute - # is still shared — if session failures return, check contention here first - # (the old shared-model setup measured 19-52s under 3x concurrency). - # Community GGUF (no first-party one) — mykor's is the most-downloaded conversion. - # Pinned to a repo revision (immutable bytes — community GGUFs can be rebuilt - # in place under the same filename); Renovate bumps the sha via the git-refs - # custom manager in .renovaterc.json5 when upstream main moves. + # harrier-oss-v1-270m Q8_0, 640-dim, 32K ctx — dedicated to vmcp-unified tool + # search so its ~194-tool batch doesn't queue behind memini on qwen3-embedding. + # Shares one iGPU with qwen3-embedding (292MB + 639MB; the device plugin + # advertises count: 2 for exactly this pair), so raw compute is still shared: + # check contention here first if session failures return. + # Community GGUF pinned to a repo revision — the same filename can be rebuilt + # in place. Renovate git-refs manager bumps the sha. source: https://huggingface.co/mykor/harrier-oss-v1-270m-GGUF/resolve/fe07a2a15994730f1b8f13943927bc8198bd7fa7/harrier-oss-v1-270M-Q8_0.gguf quantization: Q8_0 # Re-fetch when the source changes. Without this, a source change keeps @@ -40,6 +31,7 @@ spec: # nodeSelector below keeps this off the dGPU node. resourceName: squat.ai/dri --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/inference.llmkube.dev/inferenceservice_v1alpha1.json apiVersion: inference.llmkube.dev/v1alpha1 kind: InferenceService metadata: @@ -51,16 +43,11 @@ spec: # startup and model load before this Vulkan embedder is recycled. maxPodLifetimeSeconds: 86400 image: ghcr.io/ggml-org/llama.cpp:server-vulkan@sha256:eddb40b8a348a652c8829ef036df9bbbae537755ae7026bf6d4b403840cd468e - # 2 slots (the Vulkan floor — warmup hangs at parallel=1, llama.cpp #24307): - # the iGPU is compute-bound, so extra slots add interleave, not throughput - # (measured on qwen3-embedding: 4 slots didn't move latency, Q8_0 did). Two - # slots let find_tool queries flow past an in-flight session-build batch while - # halving the UMA buffer allocation on the shared iGPU. + # Vulkan floor is 2 — warmup hangs at parallel=1 (llama.cpp #24307). The iGPU + # is compute-bound, so more slots add interleave, not throughput. parallelSlots: 2 - # 8192 tokens/slot — the real batch's largest single tool description is - # 2155 tokens under harrier's Gemma3 tokenizer (~1.85x qwen3's count for the - # same text; 2048/slot rejected it live 2026-07-02). Harrier's native window - # is 32K and KV for a 270m model is tens of MB, so headroom is cheap. + # 8192 tokens/slot; largest single tool description measured 2155 tokens under + # harrier's Gemma3 tokenizer (2048/slot rejected it live 2026-07-02). contextSize: 16384 uBatchSize: 2048 # mode: embedding makes the operator inject --embedding --pooling last and @@ -74,7 +61,6 @@ spec: nodeSelector: amd.com/igpu: "true" - # Render/video host groups for /dev/dri — the documented Vulkan-runtime path. podSecurityContext: runAsUser: 0 runAsGroup: 0 @@ -82,11 +68,8 @@ spec: - 44 - 226 - # Co-locate with qwen3-embedding's pod (same node, whichever iGPU node that - # currently is — qwen3-embedding itself isn't hard-pinned to one, see - # qwen3-embedding.yaml): the two embedders deliberately share that node's iGPU - # (one squat.ai/dri slot each of the node's 2), and same-node networking for - # litellm/unified comes along for free. + # Co-locate with qwen3-embedding: the two embedders share that node's iGPU, + # one squat.ai/dri slot each of the node's 2. affinity: podAffinity: requiredDuringSchedulingIgnoredDuringExecution: @@ -95,9 +78,8 @@ spec: app: qwen3-embedding topologyKey: kubernetes.io/hostname - # No probeOverrides: the operator's llamacpp defaults already probe /health - # with a more forgiving startup window (180x10s). GPU allocation comes from - # Model.hardware.gpu.count (resources.gpu/gpuMemory are dead fields). + # GPU allocation comes from Model.hardware.gpu.count; resources.gpu/gpuMemory + # are dead fields here. resources: cpu: "500m" # inference is on the iGPU; CPU only tokenises/schedules memory: 512Mi # 292MB Q8_0 weights via UMA; measured 436Mi working set diff --git a/kubernetes/apps/ai/memini/ks.yaml b/kubernetes/apps/ai/memini/ks.yaml index 3007beaf6d..f21e21560d 100644 --- a/kubernetes/apps/ai/memini/ks.yaml +++ b/kubernetes/apps/ai/memini/ks.yaml @@ -8,8 +8,7 @@ metadata: spec: interval: 30m dependsOn: - - name: litellm - - name: llmkube-models + - name: litellm # transitively gates on llmkube-models - name: cloudnative-pg-cluster namespace: database path: ./kubernetes/apps/ai/memini/app diff --git a/kubernetes/apps/ai/odysseus/app/helmrelease.yaml b/kubernetes/apps/ai/odysseus/app/helmrelease.yaml index f8909db260..57a02f3cdd 100644 --- a/kubernetes/apps/ai/odysseus/app/helmrelease.yaml +++ b/kubernetes/apps/ai/odysseus/app/helmrelease.yaml @@ -71,7 +71,7 @@ spec: gethomepage.dev/group: AI gethomepage.dev/icon: odysseus.png hostnames: - - "odysseus.${SECRET_DOMAIN}" + - "{{ .Release.Name }}.${SECRET_DOMAIN}" parentRefs: - name: envoy-internal namespace: network diff --git a/kubernetes/apps/ai/odysseus/app/kustomization.yaml b/kubernetes/apps/ai/odysseus/app/kustomization.yaml index 4faaee06e0..904bfa2497 100644 --- a/kubernetes/apps/ai/odysseus/app/kustomization.yaml +++ b/kubernetes/apps/ai/odysseus/app/kustomization.yaml @@ -1,5 +1,5 @@ --- -# yaml-language-server: $schema=https://raw.githubusercontent.com/SchemaStore/schemastore/master/src/schemas/json/kustomization.json +# yaml-language-server: $schema=https://json.schemastore.org/kustomization apiVersion: kustomize.config.k8s.io/v1beta1 kind: Kustomization resources: diff --git a/kubernetes/apps/ai/opencode/app/helmrelease.yaml b/kubernetes/apps/ai/opencode/app/helmrelease.yaml index fd0ec8a1a5..c026f34e7d 100644 --- a/kubernetes/apps/ai/opencode/app/helmrelease.yaml +++ b/kubernetes/apps/ai/opencode/app/helmrelease.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/helm.toolkit.fluxcd.io/helmrelease_v2.json apiVersion: helm.toolkit.fluxcd.io/v2 kind: HelmRelease metadata: @@ -76,7 +77,7 @@ spec: gethomepage.dev/group: Agents gethomepage.dev/icon: opencode.png hostnames: - - "opencode.${SECRET_DOMAIN}" + - "{{ .Release.Name }}.${SECRET_DOMAIN}" parentRefs: - name: envoy-internal namespace: network diff --git a/kubernetes/apps/ai/opencode/app/opencode-vm-cors.yaml b/kubernetes/apps/ai/opencode/app/opencode-vm-cors.yaml index 85c07ade9b..45c426bb97 100644 --- a/kubernetes/apps/ai/opencode/app/opencode-vm-cors.yaml +++ b/kubernetes/apps/ai/opencode/app/opencode-vm-cors.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.envoyproxy.io/securitypolicy_v1alpha1.json apiVersion: gateway.envoyproxy.io/v1alpha1 kind: SecurityPolicy metadata: diff --git a/kubernetes/apps/ai/opencode/app/opencode-vm-endpointslice.yaml b/kubernetes/apps/ai/opencode/app/opencode-vm-endpointslice.yaml index 1c2ba4d7fb..5224f3396d 100644 --- a/kubernetes/apps/ai/opencode/app/opencode-vm-endpointslice.yaml +++ b/kubernetes/apps/ai/opencode/app/opencode-vm-endpointslice.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/discovery.k8s.io/endpointslice_v1.json apiVersion: discovery.k8s.io/v1 kind: EndpointSlice metadata: diff --git a/kubernetes/apps/ai/opencode/app/opencode-vm-httproute.yaml b/kubernetes/apps/ai/opencode/app/opencode-vm-httproute.yaml index dd25ac73db..d7fd0c91f5 100644 --- a/kubernetes/apps/ai/opencode/app/opencode-vm-httproute.yaml +++ b/kubernetes/apps/ai/opencode/app/opencode-vm-httproute.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json apiVersion: gateway.networking.k8s.io/v1 kind: HTTPRoute metadata: diff --git a/kubernetes/apps/ai/toolhive/app/helmrelease.yaml b/kubernetes/apps/ai/toolhive/app/helmrelease.yaml index 892d33a24d..d82801c50a 100644 --- a/kubernetes/apps/ai/toolhive/app/helmrelease.yaml +++ b/kubernetes/apps/ai/toolhive/app/helmrelease.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/helm.toolkit.fluxcd.io/helmrelease_v2.json apiVersion: helm.toolkit.fluxcd.io/v2 kind: HelmRelease metadata: diff --git a/kubernetes/apps/ai/toolhive/app/httproute.yaml b/kubernetes/apps/ai/toolhive/app/httproute.yaml index 0bbe8d1f88..2f505e0d00 100644 --- a/kubernetes/apps/ai/toolhive/app/httproute.yaml +++ b/kubernetes/apps/ai/toolhive/app/httproute.yaml @@ -1,161 +1,119 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json apiVersion: gateway.networking.k8s.io/v1 kind: HTTPRoute metadata: name: mcp-flux - namespace: ai spec: hostnames: - mcp-flux.${SECRET_DOMAIN} parentRefs: - - group: gateway.networking.k8s.io - kind: Gateway - name: envoy-internal + - name: envoy-internal namespace: network sectionName: https rules: - backendRefs: - name: vmcp-flux port: 4483 - matches: - - path: - type: PathPrefix - value: / --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json apiVersion: gateway.networking.k8s.io/v1 kind: HTTPRoute metadata: name: mcp-observability - namespace: ai spec: hostnames: - mcp-observability.${SECRET_DOMAIN} parentRefs: - - group: gateway.networking.k8s.io - kind: Gateway - name: envoy-internal + - name: envoy-internal namespace: network sectionName: https rules: - backendRefs: - name: vmcp-observability port: 4483 - matches: - - path: - type: PathPrefix - value: / --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json apiVersion: gateway.networking.k8s.io/v1 kind: HTTPRoute metadata: name: mcp-homeassistant - namespace: ai spec: hostnames: - mcp-homeassistant.${SECRET_DOMAIN} parentRefs: - - group: gateway.networking.k8s.io - kind: Gateway - name: envoy-internal + - name: envoy-internal namespace: network sectionName: https rules: - backendRefs: - name: vmcp-ha port: 4483 - matches: - - path: - type: PathPrefix - value: / --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json apiVersion: gateway.networking.k8s.io/v1 kind: HTTPRoute metadata: name: mcp-search - namespace: ai spec: hostnames: - mcp-search.${SECRET_DOMAIN} parentRefs: - - group: gateway.networking.k8s.io - kind: Gateway - name: envoy-internal + - name: envoy-internal namespace: network sectionName: https rules: - backendRefs: - name: vmcp-search port: 4483 - matches: - - path: - type: PathPrefix - value: / --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json apiVersion: gateway.networking.k8s.io/v1 kind: HTTPRoute metadata: name: mcp-resources - namespace: ai spec: hostnames: - mcp-resources.${SECRET_DOMAIN} parentRefs: - - group: gateway.networking.k8s.io - kind: Gateway - name: envoy-internal + - name: envoy-internal namespace: network sectionName: https rules: - backendRefs: - name: vmcp-resources port: 4483 - matches: - - path: - type: PathPrefix - value: / --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json apiVersion: gateway.networking.k8s.io/v1 kind: HTTPRoute metadata: name: mcp-database - namespace: ai spec: hostnames: - mcp-database.${SECRET_DOMAIN} parentRefs: - - group: gateway.networking.k8s.io - kind: Gateway - name: envoy-internal + - name: envoy-internal namespace: network sectionName: https rules: - backendRefs: - name: vmcp-database port: 4483 - matches: - - path: - type: PathPrefix - value: / --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json apiVersion: gateway.networking.k8s.io/v1 kind: HTTPRoute metadata: name: mcp-unified - namespace: ai spec: hostnames: - mcp.${SECRET_DOMAIN} parentRefs: - - group: gateway.networking.k8s.io - kind: Gateway - name: envoy-internal + - name: envoy-internal namespace: network sectionName: https rules: - backendRefs: - name: vmcp-unified port: 4483 - matches: - - path: - type: PathPrefix - value: / diff --git a/kubernetes/apps/ai/toolhive/app/ocirepository.yaml b/kubernetes/apps/ai/toolhive/app/ocirepository.yaml index ce43401831..8d4fcd5e05 100644 --- a/kubernetes/apps/ai/toolhive/app/ocirepository.yaml +++ b/kubernetes/apps/ai/toolhive/app/ocirepository.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/source.toolkit.fluxcd.io/ocirepository_v1.json apiVersion: source.toolkit.fluxcd.io/v1 kind: OCIRepository metadata: diff --git a/kubernetes/apps/ai/toolhive/config/context7.yaml b/kubernetes/apps/ai/toolhive/config/context7.yaml index 35240d2a32..7a873b338d 100644 --- a/kubernetes/apps/ai/toolhive/config/context7.yaml +++ b/kubernetes/apps/ai/toolhive/config/context7.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserverentry_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServerEntry metadata: diff --git a/kubernetes/apps/ai/toolhive/config/flux-operator.yaml b/kubernetes/apps/ai/toolhive/config/flux-operator.yaml index 8c8e064832..0375e8d7a4 100644 --- a/kubernetes/apps/ai/toolhive/config/flux-operator.yaml +++ b/kubernetes/apps/ai/toolhive/config/flux-operator.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -45,6 +46,7 @@ spec: cpu: 500m memory: 200Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -91,6 +93,7 @@ spec: cpu: 500m memory: 200Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcptoolconfig_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPToolConfig metadata: diff --git a/kubernetes/apps/ai/toolhive/config/github.yaml b/kubernetes/apps/ai/toolhive/config/github.yaml index e31e8f0bf7..eba6a941dd 100644 --- a/kubernetes/apps/ai/toolhive/config/github.yaml +++ b/kubernetes/apps/ai/toolhive/config/github.yaml @@ -1,3 +1,5 @@ +--- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcptoolconfig_v1beta1.json # Disabled: ToolHive v0.40.1 virtual-MCP health monitoring repeatedly initializes # GitHub's stdio backend, which returns `duplicate "initialize" received` and # degrades the resources and unified virtual MCPs. Restore after an upstream fix. @@ -74,6 +76,7 @@ # cpu: 200m # memory: 200Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcptoolconfig_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPToolConfig metadata: diff --git a/kubernetes/apps/ai/toolhive/config/grafana.yaml b/kubernetes/apps/ai/toolhive/config/grafana.yaml index bd65d4a679..9274dcb280 100644 --- a/kubernetes/apps/ai/toolhive/config/grafana.yaml +++ b/kubernetes/apps/ai/toolhive/config/grafana.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -35,6 +36,7 @@ spec: cpu: 200m memory: 100Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -71,15 +73,14 @@ spec: cpu: 200m memory: 100Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcptoolconfig_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPToolConfig metadata: name: grafana spec: - # Self-hosted Grafana OSS with Prometheus + Alertmanager only. - # Excludes Grafana Cloud features (Incident, OnCall, Sift, Assertions), - # Pyroscope tools (no Pyroscope datasource), Loki tools (no Loki datasource), - # and Sift-backed helpers (find_error_pattern_logs, find_slow_requests). + # Grafana OSS + Prometheus/Alertmanager only; Cloud, Pyroscope, Loki and + # Sift-backed tools have no datasource here. toolsFilter: # Datasources - list_datasources diff --git a/kubernetes/apps/ai/toolhive/config/homeassistant.yaml b/kubernetes/apps/ai/toolhive/config/homeassistant.yaml index 73d9aab6c3..6f5dc0e4eb 100644 --- a/kubernetes/apps/ai/toolhive/config/homeassistant.yaml +++ b/kubernetes/apps/ai/toolhive/config/homeassistant.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -62,6 +63,7 @@ spec: cpu: 200m memory: 128Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -117,7 +119,6 @@ spec: memory: 64Mi limits: cpu: 1 - # 13 OOMKills in 30d at 256Mi; observed peak 255Mi on both servers. memory: 512Mi resources: requests: @@ -127,6 +128,7 @@ spec: cpu: 200m memory: 128Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcptoolconfig_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPToolConfig metadata: diff --git a/kubernetes/apps/ai/toolhive/config/karakeep.yaml b/kubernetes/apps/ai/toolhive/config/karakeep.yaml index 87754901e8..47dc6e6909 100644 --- a/kubernetes/apps/ai/toolhive/config/karakeep.yaml +++ b/kubernetes/apps/ai/toolhive/config/karakeep.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -36,6 +37,7 @@ spec: cpu: 200m memory: 100Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -63,7 +65,6 @@ spec: memory: 64Mi limits: cpu: 500m - # Peaks at 84Mi, i.e. 84% of the old 100Mi limit. memory: 256Mi resources: requests: @@ -73,6 +74,7 @@ spec: cpu: 200m memory: 100Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcptoolconfig_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPToolConfig metadata: diff --git a/kubernetes/apps/ai/toolhive/config/kubesearch.yaml b/kubernetes/apps/ai/toolhive/config/kubesearch.yaml index 259cd61e6f..fa20c8b402 100644 --- a/kubernetes/apps/ai/toolhive/config/kubesearch.yaml +++ b/kubernetes/apps/ai/toolhive/config/kubesearch.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -55,6 +56,7 @@ spec: cpu: 200m memory: 256Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -111,6 +113,7 @@ spec: cpu: 200m memory: 256Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcptoolconfig_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPToolConfig metadata: diff --git a/kubernetes/apps/ai/toolhive/config/mcpgroups.yaml b/kubernetes/apps/ai/toolhive/config/mcpgroups.yaml index 97300a5a55..4d3a8da2a5 100644 --- a/kubernetes/apps/ai/toolhive/config/mcpgroups.yaml +++ b/kubernetes/apps/ai/toolhive/config/mcpgroups.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpgroup_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPGroup metadata: @@ -6,6 +7,7 @@ metadata: spec: description: Flux reconcile, K8s/operator tasks --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpgroup_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPGroup metadata: @@ -13,6 +15,7 @@ metadata: spec: description: Dashboards, alerts --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpgroup_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPGroup metadata: @@ -20,6 +23,7 @@ metadata: spec: description: HA controls --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpgroup_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPGroup metadata: @@ -27,6 +31,7 @@ metadata: spec: description: Web search --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpgroup_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPGroup metadata: @@ -34,6 +39,7 @@ metadata: spec: description: Postgres health and tuning (read-only) --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpgroup_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPGroup metadata: @@ -41,9 +47,10 @@ metadata: spec: description: Repos, bookmarks --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpgroup_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPGroup metadata: name: all spec: - description: Optimizer test — duplicate backends aggregating all tools + description: Optimizer backends — all tools aggregated diff --git a/kubernetes/apps/ai/toolhive/config/postgres-mcp.yaml b/kubernetes/apps/ai/toolhive/config/postgres-mcp.yaml index bc3a1cecc6..c1dccde55c 100644 --- a/kubernetes/apps/ai/toolhive/config/postgres-mcp.yaml +++ b/kubernetes/apps/ai/toolhive/config/postgres-mcp.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -35,6 +36,7 @@ spec: cpu: 500m memory: 512Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -71,6 +73,7 @@ spec: cpu: 500m memory: 512Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcptoolconfig_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPToolConfig metadata: diff --git a/kubernetes/apps/ai/toolhive/config/searxng.yaml b/kubernetes/apps/ai/toolhive/config/searxng.yaml index e82217420b..68f3086c81 100644 --- a/kubernetes/apps/ai/toolhive/config/searxng.yaml +++ b/kubernetes/apps/ai/toolhive/config/searxng.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -9,7 +10,6 @@ spec: groupRef: name: search env: - # Service in default namespace exposes port 8080 (see searxng HelmRelease). - name: SEARXNG_URL value: http://searxng.default.svc.cluster.local:8080 toolConfigRef: @@ -33,6 +33,7 @@ spec: cpu: 200m memory: 100Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -43,7 +44,6 @@ spec: groupRef: name: all env: - # Service in default namespace exposes port 8080 (see searxng HelmRelease). - name: SEARXNG_URL value: http://searxng.default.svc.cluster.local:8080 toolConfigRef: @@ -67,6 +67,7 @@ spec: cpu: 200m memory: 100Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcptoolconfig_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPToolConfig metadata: diff --git a/kubernetes/apps/ai/toolhive/config/talos-mcp.yaml b/kubernetes/apps/ai/toolhive/config/talos-mcp.yaml index 6233077826..816198f392 100644 --- a/kubernetes/apps/ai/toolhive/config/talos-mcp.yaml +++ b/kubernetes/apps/ai/toolhive/config/talos-mcp.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/talos.dev/serviceaccount_v1alpha1.json apiVersion: talos.dev/v1alpha1 kind: ServiceAccount metadata: @@ -6,6 +7,7 @@ metadata: spec: roles: ["os:reader"] --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -59,6 +61,7 @@ spec: cpu: 200m memory: 128Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: @@ -112,6 +115,7 @@ spec: cpu: 200m memory: 128Mi --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcptoolconfig_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPToolConfig metadata: diff --git a/kubernetes/apps/ai/toolhive/config/virtualmcpservers.yaml b/kubernetes/apps/ai/toolhive/config/virtualmcpservers.yaml index 37c09ed5f3..01583f4ea7 100644 --- a/kubernetes/apps/ai/toolhive/config/virtualmcpservers.yaml +++ b/kubernetes/apps/ai/toolhive/config/virtualmcpservers.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/virtualmcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: VirtualMCPServer metadata: @@ -8,11 +9,9 @@ spec: sessionStorage: provider: redis address: dragonfly.database.svc.cluster.local:6379 - db: 0 keyPrefix: "thv:vmcp:flux:session:" incomingAuth: type: anonymous - serviceType: ClusterIP groupRef: name: flux config: @@ -21,6 +20,7 @@ spec: conflictResolutionConfig: prefixFormat: "{workload}_" --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/virtualmcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: VirtualMCPServer metadata: @@ -42,11 +42,9 @@ spec: sessionStorage: provider: redis address: dragonfly.database.svc.cluster.local:6379 - db: 0 keyPrefix: "thv:vmcp:observability:session:" incomingAuth: type: anonymous - serviceType: ClusterIP groupRef: name: observability config: @@ -55,6 +53,7 @@ spec: conflictResolutionConfig: prefixFormat: "{workload}_" --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/virtualmcpserver_v1beta1.json # Named "ha" to avoid Deployment name collision with MCPServer "homeassistant" # (operator creates Deployment = name). apiVersion: toolhive.stacklok.dev/v1beta1 @@ -78,11 +77,9 @@ spec: sessionStorage: provider: redis address: dragonfly.database.svc.cluster.local:6379 - db: 0 keyPrefix: "thv:vmcp:ha:session:" incomingAuth: type: anonymous - serviceType: ClusterIP groupRef: name: homeassistant config: @@ -91,6 +88,7 @@ spec: conflictResolutionConfig: prefixFormat: "{workload}_" --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/virtualmcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: VirtualMCPServer metadata: @@ -100,11 +98,9 @@ spec: sessionStorage: provider: redis address: dragonfly.database.svc.cluster.local:6379 - db: 0 keyPrefix: "thv:vmcp:search:session:" incomingAuth: type: anonymous - serviceType: ClusterIP groupRef: name: search config: @@ -113,6 +109,7 @@ spec: conflictResolutionConfig: prefixFormat: "{workload}_" --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/virtualmcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: VirtualMCPServer metadata: @@ -122,11 +119,9 @@ spec: sessionStorage: provider: redis address: dragonfly.database.svc.cluster.local:6379 - db: 0 keyPrefix: "thv:vmcp:database:session:" incomingAuth: type: anonymous - serviceType: ClusterIP groupRef: name: database config: @@ -135,6 +130,7 @@ spec: conflictResolutionConfig: prefixFormat: "{workload}_" --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/virtualmcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: VirtualMCPServer metadata: @@ -144,11 +140,9 @@ spec: sessionStorage: provider: redis address: dragonfly.database.svc.cluster.local:6379 - db: 0 keyPrefix: "thv:vmcp:resources:session:" incomingAuth: type: anonymous - serviceType: ClusterIP groupRef: name: resources config: @@ -157,6 +151,7 @@ spec: conflictResolutionConfig: prefixFormat: "{workload}_" --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/virtualmcpserver_v1beta1.json apiVersion: toolhive.stacklok.dev/v1beta1 kind: VirtualMCPServer metadata: @@ -207,23 +202,16 @@ spec: sessionStorage: provider: redis address: dragonfly.database.svc.cluster.local:6379 - db: 0 keyPrefix: "thv:vmcp:unified:session:" incomingAuth: type: anonymous - serviceType: ClusterIP groupRef: name: all config: - # Session build re-embeds all ~194 tools on initialize; litellm's embedding - # cache (kubernetes/apps/ai/litellm/instance/proxy.yaml) makes this a dragonfly - # hit in the common case, but a cold cache still takes ~27-35s on the iGPU, - # which blew past the 30s operational default (2026-07-03). Kept in sync with - # optimizer.embeddingServiceTimeout via anchor since both bound the same call. - # ponytail: `default` applies to ALL backend requests, not just session-build, - # so every other call now tolerates hangs up to 120s too. perWorkload could - # scope this narrower, but the slow leg is the embedding call (not a specific - # backend workload) — revisit if a hung backend's slow-failure becomes a problem. + # Cold embedding cache takes ~27-35s on the iGPU, past the 30s default + # (2026-07-03). Anchored to embeddingServiceTimeout — same call. + # ponytail: `default` covers every backend request, not just session-build; + # switch to perWorkload if a hung backend's slow failure starts mattering. operational: timeouts: default: &embeddingTimeout 120s diff --git a/kubernetes/apps/ai/toolhive/crds/helmrelease.yaml b/kubernetes/apps/ai/toolhive/crds/helmrelease.yaml index 70ef64eb9d..6f7a675c89 100644 --- a/kubernetes/apps/ai/toolhive/crds/helmrelease.yaml +++ b/kubernetes/apps/ai/toolhive/crds/helmrelease.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/helm.toolkit.fluxcd.io/helmrelease_v2.json apiVersion: helm.toolkit.fluxcd.io/v2 kind: HelmRelease metadata: diff --git a/kubernetes/apps/ai/toolhive/crds/ocirepository.yaml b/kubernetes/apps/ai/toolhive/crds/ocirepository.yaml index ab759710fe..dccfd84e4a 100644 --- a/kubernetes/apps/ai/toolhive/crds/ocirepository.yaml +++ b/kubernetes/apps/ai/toolhive/crds/ocirepository.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/source.toolkit.fluxcd.io/ocirepository_v1.json apiVersion: source.toolkit.fluxcd.io/v1 kind: OCIRepository metadata: diff --git a/kubernetes/apps/ai/toolhive/ks.yaml b/kubernetes/apps/ai/toolhive/ks.yaml index b60e5c8126..e8e205b405 100644 --- a/kubernetes/apps/ai/toolhive/ks.yaml +++ b/kubernetes/apps/ai/toolhive/ks.yaml @@ -1,9 +1,10 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json apiVersion: kustomize.toolkit.fluxcd.io/v1 kind: Kustomization metadata: name: toolhive-operator-crds - namespace: ai + namespace: &namespace ai spec: commonMetadata: labels: @@ -15,13 +16,14 @@ spec: kind: GitRepository name: flux-system namespace: flux-system - targetNamespace: ai + targetNamespace: *namespace --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json apiVersion: kustomize.toolkit.fluxcd.io/v1 kind: Kustomization metadata: name: toolhive-operator - namespace: ai + namespace: &namespace ai spec: commonMetadata: labels: @@ -35,17 +37,18 @@ spec: kind: GitRepository name: flux-system namespace: flux-system - targetNamespace: ai + targetNamespace: *namespace --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json apiVersion: kustomize.toolkit.fluxcd.io/v1 kind: Kustomization metadata: - name: toolhive - namespace: ai + name: &app toolhive + namespace: &namespace ai spec: commonMetadata: labels: - app.kubernetes.io/name: toolhive + app.kubernetes.io/name: *app dependsOn: - name: toolhive-operator interval: 1h @@ -55,4 +58,4 @@ spec: kind: GitRepository name: flux-system namespace: flux-system - targetNamespace: ai + targetNamespace: *namespace diff --git a/kubernetes/apps/database/cloudnative-pg/barman-cloud/helmrelease.yaml b/kubernetes/apps/database/cloudnative-pg/barman-cloud/helmrelease.yaml index 4dd44a5b7a..500b9055bf 100644 --- a/kubernetes/apps/database/cloudnative-pg/barman-cloud/helmrelease.yaml +++ b/kubernetes/apps/database/cloudnative-pg/barman-cloud/helmrelease.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/helm.toolkit.fluxcd.io/helmrelease_v2.json apiVersion: helm.toolkit.fluxcd.io/v2 kind: HelmRelease metadata: diff --git a/kubernetes/apps/database/cloudnative-pg/cluster/pooler-ro.yaml b/kubernetes/apps/database/cloudnative-pg/cluster/pooler-ro.yaml index be07183446..a41e360f19 100644 --- a/kubernetes/apps/database/cloudnative-pg/cluster/pooler-ro.yaml +++ b/kubernetes/apps/database/cloudnative-pg/cluster/pooler-ro.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/postgresql.cnpg.io/pooler_v1.json # Transaction-mode RO PgBouncer → replicas only (no primary max_connections usage). # Per logical DB: up to instances × max_db_connections server conns from this deployment. apiVersion: postgresql.cnpg.io/v1 diff --git a/kubernetes/apps/database/cloudnative-pg/cluster/pooler-session.yaml b/kubernetes/apps/database/cloudnative-pg/cluster/pooler-session.yaml index dd8505e933..bb4496eec0 100644 --- a/kubernetes/apps/database/cloudnative-pg/cluster/pooler-session.yaml +++ b/kubernetes/apps/database/cloudnative-pg/cluster/pooler-session.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/postgresql.cnpg.io/pooler_v1.json # Session-mode RW pooler (prepared statements, advisory locks, etc.). Service: # pgbouncer-session.database.svc.cluster.local:5432 — same primary as pgbouncer-rw. apiVersion: postgresql.cnpg.io/v1 diff --git a/kubernetes/apps/database/cloudnative-pg/cluster/pooler.yaml b/kubernetes/apps/database/cloudnative-pg/cluster/pooler.yaml index a8c2dd23e9..b85a41202f 100644 --- a/kubernetes/apps/database/cloudnative-pg/cluster/pooler.yaml +++ b/kubernetes/apps/database/cloudnative-pg/cluster/pooler.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/postgresql.cnpg.io/pooler_v1.json # Transaction-mode RW PgBouncer → primary. Per pod, max_db_connections caps server conns # to one Postgres database (PgBouncer docs). Through this Pooler only: up to # instances × max_db_connections per DB; multiple DBs on the primary add separately. diff --git a/kubernetes/apps/database/dragonfly/cluster/pdb.yaml b/kubernetes/apps/database/dragonfly/cluster/pdb.yaml index f4b5999055..4b75ac04e4 100644 --- a/kubernetes/apps/database/dragonfly/cluster/pdb.yaml +++ b/kubernetes/apps/database/dragonfly/cluster/pdb.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/policy/poddisruptionbudget_v1.json apiVersion: policy/v1 kind: PodDisruptionBudget metadata: diff --git a/kubernetes/apps/default/immich/machine-learning/kustomization.yaml b/kubernetes/apps/default/immich/machine-learning/kustomization.yaml index c4c1dfd418..f57d1f2315 100644 --- a/kubernetes/apps/default/immich/machine-learning/kustomization.yaml +++ b/kubernetes/apps/default/immich/machine-learning/kustomization.yaml @@ -1,5 +1,5 @@ --- -# yaml-language-server: $schema=https://raw.githubusercontent.com/SchemaStore/schemastore/master/src/schemas/json/kustomization.json +# yaml-language-server: $schema=https://json.schemastore.org/kustomization apiVersion: kustomize.config.k8s.io/v1beta1 kind: Kustomization namespace: default diff --git a/kubernetes/apps/default/immich/server/kustomization.yaml b/kubernetes/apps/default/immich/server/kustomization.yaml index ce2191c7db..12ce2e1b4c 100644 --- a/kubernetes/apps/default/immich/server/kustomization.yaml +++ b/kubernetes/apps/default/immich/server/kustomization.yaml @@ -1,5 +1,5 @@ --- -# yaml-language-server: $schema=https://raw.githubusercontent.com/SchemaStore/schemastore/master/src/schemas/json/kustomization.json +# yaml-language-server: $schema=https://json.schemastore.org/kustomization apiVersion: kustomize.config.k8s.io/v1beta1 kind: Kustomization namespace: default diff --git a/kubernetes/apps/default/nextcloud/app/kustomization.yaml b/kubernetes/apps/default/nextcloud/app/kustomization.yaml index 892a1d722a..151536e6a4 100644 --- a/kubernetes/apps/default/nextcloud/app/kustomization.yaml +++ b/kubernetes/apps/default/nextcloud/app/kustomization.yaml @@ -1,5 +1,5 @@ --- -# yaml-language-server: $schema=https://raw.githubusercontent.com/SchemaStore/schemastore/master/src/schemas/json/kustomization.json +# yaml-language-server: $schema=https://json.schemastore.org/kustomization apiVersion: kustomize.config.k8s.io/v1beta1 kind: Kustomization resources: diff --git a/kubernetes/apps/default/nextcloud/facerecognition/kustomization.yaml b/kubernetes/apps/default/nextcloud/facerecognition/kustomization.yaml index 214423845d..69bbe09e13 100644 --- a/kubernetes/apps/default/nextcloud/facerecognition/kustomization.yaml +++ b/kubernetes/apps/default/nextcloud/facerecognition/kustomization.yaml @@ -1,5 +1,5 @@ --- -# yaml-language-server: $schema=https://raw.githubusercontent.com/SchemaStore/schemastore/master/src/schemas/json/kustomization.json +# yaml-language-server: $schema=https://json.schemastore.org/kustomization apiVersion: kustomize.config.k8s.io/v1beta1 kind: Kustomization resources: diff --git a/kubernetes/apps/default/nextcloud/notify-push/kustomization.yaml b/kubernetes/apps/default/nextcloud/notify-push/kustomization.yaml index 4faaee06e0..904bfa2497 100644 --- a/kubernetes/apps/default/nextcloud/notify-push/kustomization.yaml +++ b/kubernetes/apps/default/nextcloud/notify-push/kustomization.yaml @@ -1,5 +1,5 @@ --- -# yaml-language-server: $schema=https://raw.githubusercontent.com/SchemaStore/schemastore/master/src/schemas/json/kustomization.json +# yaml-language-server: $schema=https://json.schemastore.org/kustomization apiVersion: kustomize.config.k8s.io/v1beta1 kind: Kustomization resources: diff --git a/kubernetes/apps/default/obico/app/kustomization.yaml b/kubernetes/apps/default/obico/app/kustomization.yaml index 476326bf2e..d46239ef27 100644 --- a/kubernetes/apps/default/obico/app/kustomization.yaml +++ b/kubernetes/apps/default/obico/app/kustomization.yaml @@ -1,5 +1,5 @@ --- -# yaml-language-server: $schema=https://raw.githubusercontent.com/SchemaStore/schemastore/master/src/schemas/json/kustomization.json +# yaml-language-server: $schema=https://json.schemastore.org/kustomization apiVersion: kustomize.config.k8s.io/v1beta1 kind: Kustomization resources: diff --git a/kubernetes/apps/default/picoshare/app/kustomization.yaml b/kubernetes/apps/default/picoshare/app/kustomization.yaml index 214423845d..69bbe09e13 100644 --- a/kubernetes/apps/default/picoshare/app/kustomization.yaml +++ b/kubernetes/apps/default/picoshare/app/kustomization.yaml @@ -1,5 +1,5 @@ --- -# yaml-language-server: $schema=https://raw.githubusercontent.com/SchemaStore/schemastore/master/src/schemas/json/kustomization.json +# yaml-language-server: $schema=https://json.schemastore.org/kustomization apiVersion: kustomize.config.k8s.io/v1beta1 kind: Kustomization resources: diff --git a/kubernetes/apps/default/searxng/app/pdb.yaml b/kubernetes/apps/default/searxng/app/pdb.yaml index f19aac64e8..1c42b62422 100644 --- a/kubernetes/apps/default/searxng/app/pdb.yaml +++ b/kubernetes/apps/default/searxng/app/pdb.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/policy/poddisruptionbudget_v1.json apiVersion: policy/v1 kind: PodDisruptionBudget metadata: diff --git a/kubernetes/apps/default/spoolman/app/kustomization.yaml b/kubernetes/apps/default/spoolman/app/kustomization.yaml index 214423845d..69bbe09e13 100644 --- a/kubernetes/apps/default/spoolman/app/kustomization.yaml +++ b/kubernetes/apps/default/spoolman/app/kustomization.yaml @@ -1,5 +1,5 @@ --- -# yaml-language-server: $schema=https://raw.githubusercontent.com/SchemaStore/schemastore/master/src/schemas/json/kustomization.json +# yaml-language-server: $schema=https://json.schemastore.org/kustomization apiVersion: kustomize.config.k8s.io/v1beta1 kind: Kustomization resources: diff --git a/kubernetes/apps/flux-system/flux-operator/app/clusterrolebinding.yaml b/kubernetes/apps/flux-system/flux-operator/app/clusterrolebinding.yaml index 6636ad0598..dbd53a48e1 100644 --- a/kubernetes/apps/flux-system/flux-operator/app/clusterrolebinding.yaml +++ b/kubernetes/apps/flux-system/flux-operator/app/clusterrolebinding.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/rbac.authorization.k8s.io/clusterrolebinding_v1.json apiVersion: rbac.authorization.k8s.io/v1 kind: ClusterRoleBinding metadata: diff --git a/kubernetes/apps/kube-system/amdgpu-undervolt.yaml b/kubernetes/apps/kube-system/amdgpu-undervolt.yaml index 30a1c65d62..17efabf665 100644 --- a/kubernetes/apps/kube-system/amdgpu-undervolt.yaml +++ b/kubernetes/apps/kube-system/amdgpu-undervolt.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/apps/daemonset_v1.json apiVersion: apps/v1 kind: DaemonSet metadata: diff --git a/kubernetes/apps/kube-system/priorityclass.yaml b/kubernetes/apps/kube-system/priorityclass.yaml index ca4996c491..18fac62195 100644 --- a/kubernetes/apps/kube-system/priorityclass.yaml +++ b/kubernetes/apps/kube-system/priorityclass.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/scheduling.k8s.io/priorityclass_v1.json # Unclassified pods default to 0, so this alone restores preemption ordering for vmagent/vmalert/vmauth/vmsingle after the 07-23 outage. apiVersion: scheduling.k8s.io/v1 kind: PriorityClass diff --git a/kubernetes/apps/network/cloudflared/app/pdb.yaml b/kubernetes/apps/network/cloudflared/app/pdb.yaml index 7ec4f15c46..f84bb3b9cc 100644 --- a/kubernetes/apps/network/cloudflared/app/pdb.yaml +++ b/kubernetes/apps/network/cloudflared/app/pdb.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/policy/poddisruptionbudget_v1.json apiVersion: policy/v1 kind: PodDisruptionBudget metadata: diff --git a/kubernetes/apps/network/external-service/arm/endpointslice.yaml b/kubernetes/apps/network/external-service/arm/endpointslice.yaml index 8cec20d12a..088d9c046a 100644 --- a/kubernetes/apps/network/external-service/arm/endpointslice.yaml +++ b/kubernetes/apps/network/external-service/arm/endpointslice.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/discovery.k8s.io/endpointslice_v1.json apiVersion: discovery.k8s.io/v1 kind: EndpointSlice metadata: diff --git a/kubernetes/apps/network/external-service/arm/httproute.yaml b/kubernetes/apps/network/external-service/arm/httproute.yaml index 12eae0c00e..893ebf92c0 100644 --- a/kubernetes/apps/network/external-service/arm/httproute.yaml +++ b/kubernetes/apps/network/external-service/arm/httproute.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json apiVersion: gateway.networking.k8s.io/v1 kind: HTTPRoute metadata: diff --git a/kubernetes/apps/network/external-service/avr/endpointslice.yaml b/kubernetes/apps/network/external-service/avr/endpointslice.yaml index 0ef7f8de64..915c498da3 100644 --- a/kubernetes/apps/network/external-service/avr/endpointslice.yaml +++ b/kubernetes/apps/network/external-service/avr/endpointslice.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/discovery.k8s.io/endpointslice_v1.json apiVersion: discovery.k8s.io/v1 kind: EndpointSlice metadata: diff --git a/kubernetes/apps/network/external-service/avr/httproute.yaml b/kubernetes/apps/network/external-service/avr/httproute.yaml index 6bacdb6380..7cf52b0986 100644 --- a/kubernetes/apps/network/external-service/avr/httproute.yaml +++ b/kubernetes/apps/network/external-service/avr/httproute.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json apiVersion: gateway.networking.k8s.io/v1 kind: HTTPRoute metadata: diff --git a/kubernetes/apps/network/external-service/centauri-carbon/endpointslice.yaml b/kubernetes/apps/network/external-service/centauri-carbon/endpointslice.yaml index 878fe04ae4..fc8c87dc08 100644 --- a/kubernetes/apps/network/external-service/centauri-carbon/endpointslice.yaml +++ b/kubernetes/apps/network/external-service/centauri-carbon/endpointslice.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/discovery.k8s.io/endpointslice_v1.json apiVersion: discovery.k8s.io/v1 kind: EndpointSlice metadata: diff --git a/kubernetes/apps/network/external-service/centauri-carbon/httproute.yaml b/kubernetes/apps/network/external-service/centauri-carbon/httproute.yaml index 48da04aa95..2e26019a12 100644 --- a/kubernetes/apps/network/external-service/centauri-carbon/httproute.yaml +++ b/kubernetes/apps/network/external-service/centauri-carbon/httproute.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json apiVersion: gateway.networking.k8s.io/v1 kind: HTTPRoute metadata: diff --git a/kubernetes/apps/network/external-service/homeassistant/endpointslice.yaml b/kubernetes/apps/network/external-service/homeassistant/endpointslice.yaml index 8e70360665..62af4988ea 100644 --- a/kubernetes/apps/network/external-service/homeassistant/endpointslice.yaml +++ b/kubernetes/apps/network/external-service/homeassistant/endpointslice.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/discovery.k8s.io/endpointslice_v1.json apiVersion: discovery.k8s.io/v1 kind: EndpointSlice metadata: diff --git a/kubernetes/apps/network/external-service/homeassistant/httproute.yaml b/kubernetes/apps/network/external-service/homeassistant/httproute.yaml index cf4f050898..db77ff3a2d 100644 --- a/kubernetes/apps/network/external-service/homeassistant/httproute.yaml +++ b/kubernetes/apps/network/external-service/homeassistant/httproute.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json apiVersion: gateway.networking.k8s.io/v1 kind: HTTPRoute metadata: diff --git a/kubernetes/apps/network/external-service/ipmi/endpointslice.yaml b/kubernetes/apps/network/external-service/ipmi/endpointslice.yaml index 9d8d5075af..83d312b526 100644 --- a/kubernetes/apps/network/external-service/ipmi/endpointslice.yaml +++ b/kubernetes/apps/network/external-service/ipmi/endpointslice.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/discovery.k8s.io/endpointslice_v1.json apiVersion: discovery.k8s.io/v1 kind: EndpointSlice metadata: diff --git a/kubernetes/apps/network/external-service/ipmi/tlsroute.yaml b/kubernetes/apps/network/external-service/ipmi/tlsroute.yaml index 44eb353ef0..cc8459725a 100644 --- a/kubernetes/apps/network/external-service/ipmi/tlsroute.yaml +++ b/kubernetes/apps/network/external-service/ipmi/tlsroute.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/tlsroute_v1alpha2.json apiVersion: gateway.networking.k8s.io/v1alpha2 kind: TLSRoute metadata: diff --git a/kubernetes/apps/network/external-service/ntopg/endpointslice.yaml b/kubernetes/apps/network/external-service/ntopg/endpointslice.yaml index ceddbd682f..a118d45f03 100644 --- a/kubernetes/apps/network/external-service/ntopg/endpointslice.yaml +++ b/kubernetes/apps/network/external-service/ntopg/endpointslice.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/discovery.k8s.io/endpointslice_v1.json apiVersion: discovery.k8s.io/v1 kind: EndpointSlice metadata: diff --git a/kubernetes/apps/network/external-service/ntopg/httproute.yaml b/kubernetes/apps/network/external-service/ntopg/httproute.yaml index 7cf586debb..8042bf3322 100644 --- a/kubernetes/apps/network/external-service/ntopg/httproute.yaml +++ b/kubernetes/apps/network/external-service/ntopg/httproute.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json apiVersion: gateway.networking.k8s.io/v1 kind: HTTPRoute metadata: diff --git a/kubernetes/apps/network/external-service/opnsense/endpointslice.yaml b/kubernetes/apps/network/external-service/opnsense/endpointslice.yaml index b94c31681c..32c7d2adf6 100644 --- a/kubernetes/apps/network/external-service/opnsense/endpointslice.yaml +++ b/kubernetes/apps/network/external-service/opnsense/endpointslice.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/discovery.k8s.io/endpointslice_v1.json apiVersion: discovery.k8s.io/v1 kind: EndpointSlice metadata: diff --git a/kubernetes/apps/network/external-service/opnsense/tlsroute.yaml b/kubernetes/apps/network/external-service/opnsense/tlsroute.yaml index bdd3ccbdcd..58e21a97b6 100644 --- a/kubernetes/apps/network/external-service/opnsense/tlsroute.yaml +++ b/kubernetes/apps/network/external-service/opnsense/tlsroute.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/tlsroute_v1alpha2.json apiVersion: gateway.networking.k8s.io/v1alpha2 kind: TLSRoute metadata: diff --git a/kubernetes/apps/network/external-service/scrutiny/endpointslice.yaml b/kubernetes/apps/network/external-service/scrutiny/endpointslice.yaml index 9d640d136c..edd5d87f26 100644 --- a/kubernetes/apps/network/external-service/scrutiny/endpointslice.yaml +++ b/kubernetes/apps/network/external-service/scrutiny/endpointslice.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/discovery.k8s.io/endpointslice_v1.json apiVersion: discovery.k8s.io/v1 kind: EndpointSlice metadata: diff --git a/kubernetes/apps/network/external-service/scrutiny/httproute.yaml b/kubernetes/apps/network/external-service/scrutiny/httproute.yaml index b24723c6ea..f7e8e4671c 100644 --- a/kubernetes/apps/network/external-service/scrutiny/httproute.yaml +++ b/kubernetes/apps/network/external-service/scrutiny/httproute.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json apiVersion: gateway.networking.k8s.io/v1 kind: HTTPRoute metadata: diff --git a/kubernetes/apps/network/external-service/truenas/endpointslice.yaml b/kubernetes/apps/network/external-service/truenas/endpointslice.yaml index 40393dbfca..7292cfcebb 100644 --- a/kubernetes/apps/network/external-service/truenas/endpointslice.yaml +++ b/kubernetes/apps/network/external-service/truenas/endpointslice.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/discovery.k8s.io/endpointslice_v1.json apiVersion: discovery.k8s.io/v1 kind: EndpointSlice metadata: diff --git a/kubernetes/apps/network/external-service/truenas/tlsroute.yaml b/kubernetes/apps/network/external-service/truenas/tlsroute.yaml index 8222d1912b..46d75dc0eb 100644 --- a/kubernetes/apps/network/external-service/truenas/tlsroute.yaml +++ b/kubernetes/apps/network/external-service/truenas/tlsroute.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/tlsroute_v1alpha2.json apiVersion: gateway.networking.k8s.io/v1alpha2 kind: TLSRoute metadata: diff --git a/kubernetes/apps/network/external-service/vaultwarden/endpointslice.yaml b/kubernetes/apps/network/external-service/vaultwarden/endpointslice.yaml index 5c38eb4b2f..51cf9dd654 100644 --- a/kubernetes/apps/network/external-service/vaultwarden/endpointslice.yaml +++ b/kubernetes/apps/network/external-service/vaultwarden/endpointslice.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/discovery.k8s.io/endpointslice_v1.json apiVersion: discovery.k8s.io/v1 kind: EndpointSlice metadata: diff --git a/kubernetes/apps/network/external-service/vaultwarden/httproute.yaml b/kubernetes/apps/network/external-service/vaultwarden/httproute.yaml index bdba430c6f..8fa3bc1caa 100644 --- a/kubernetes/apps/network/external-service/vaultwarden/httproute.yaml +++ b/kubernetes/apps/network/external-service/vaultwarden/httproute.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json apiVersion: gateway.networking.k8s.io/v1 kind: HTTPRoute metadata: diff --git a/kubernetes/apps/observability/exporters/nut-exporter/ks.yaml b/kubernetes/apps/observability/exporters/nut-exporter/ks.yaml index 041cf241b0..63ea824ff5 100644 --- a/kubernetes/apps/observability/exporters/nut-exporter/ks.yaml +++ b/kubernetes/apps/observability/exporters/nut-exporter/ks.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json apiVersion: kustomize.toolkit.fluxcd.io/v1 kind: Kustomization metadata: diff --git a/kubernetes/apps/observability/exporters/prowlarr-exporter/ks.yaml b/kubernetes/apps/observability/exporters/prowlarr-exporter/ks.yaml index ff798e769e..47c4c0dfc1 100644 --- a/kubernetes/apps/observability/exporters/prowlarr-exporter/ks.yaml +++ b/kubernetes/apps/observability/exporters/prowlarr-exporter/ks.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json apiVersion: kustomize.toolkit.fluxcd.io/v1 kind: Kustomization metadata: diff --git a/kubernetes/apps/observability/exporters/qbittorrent-exporter/ks.yaml b/kubernetes/apps/observability/exporters/qbittorrent-exporter/ks.yaml index ae5f741b2c..0e8f1b754d 100644 --- a/kubernetes/apps/observability/exporters/qbittorrent-exporter/ks.yaml +++ b/kubernetes/apps/observability/exporters/qbittorrent-exporter/ks.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json apiVersion: kustomize.toolkit.fluxcd.io/v1 kind: Kustomization metadata: diff --git a/kubernetes/apps/observability/exporters/sonarr-exporter/ks.yaml b/kubernetes/apps/observability/exporters/sonarr-exporter/ks.yaml index bfe82b27cd..6050bee060 100644 --- a/kubernetes/apps/observability/exporters/sonarr-exporter/ks.yaml +++ b/kubernetes/apps/observability/exporters/sonarr-exporter/ks.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json apiVersion: kustomize.toolkit.fluxcd.io/v1 kind: Kustomization metadata: diff --git a/kubernetes/apps/observability/exporters/speedtest-exporter/ks.yaml b/kubernetes/apps/observability/exporters/speedtest-exporter/ks.yaml index 27f68513cd..c3f6ce07d1 100644 --- a/kubernetes/apps/observability/exporters/speedtest-exporter/ks.yaml +++ b/kubernetes/apps/observability/exporters/speedtest-exporter/ks.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json apiVersion: kustomize.toolkit.fluxcd.io/v1 kind: Kustomization metadata: diff --git a/kubernetes/apps/observability/kromgo/app/helmrelease.yaml b/kubernetes/apps/observability/kromgo/app/helmrelease.yaml index aa269e78f6..20e508b08f 100644 --- a/kubernetes/apps/observability/kromgo/app/helmrelease.yaml +++ b/kubernetes/apps/observability/kromgo/app/helmrelease.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/helm.toolkit.fluxcd.io/helmrelease_v2.json apiVersion: helm.toolkit.fluxcd.io/v2 kind: HelmRelease metadata: diff --git a/kubernetes/apps/observability/kromgo/app/ocirepository.yaml b/kubernetes/apps/observability/kromgo/app/ocirepository.yaml index 65416bed86..5bb6ae0d2c 100644 --- a/kubernetes/apps/observability/kromgo/app/ocirepository.yaml +++ b/kubernetes/apps/observability/kromgo/app/ocirepository.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/source.toolkit.fluxcd.io/ocirepository_v1.json apiVersion: source.toolkit.fluxcd.io/v1 kind: OCIRepository metadata: diff --git a/kubernetes/apps/observability/siren/app/helmrelease.yaml b/kubernetes/apps/observability/siren/app/helmrelease.yaml index 000cb37d97..600a55b1af 100644 --- a/kubernetes/apps/observability/siren/app/helmrelease.yaml +++ b/kubernetes/apps/observability/siren/app/helmrelease.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/helm.toolkit.fluxcd.io/helmrelease_v2.json apiVersion: helm.toolkit.fluxcd.io/v2 kind: HelmRelease metadata: diff --git a/kubernetes/apps/observability/siren/ks.yaml b/kubernetes/apps/observability/siren/ks.yaml index c4789858cd..5c850dd737 100644 --- a/kubernetes/apps/observability/siren/ks.yaml +++ b/kubernetes/apps/observability/siren/ks.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json apiVersion: kustomize.toolkit.fluxcd.io/v1 kind: Kustomization metadata: diff --git a/kubernetes/apps/security/crowdsec/bouncers/envoy/referencegrant.yaml b/kubernetes/apps/security/crowdsec/bouncers/envoy/referencegrant.yaml index cc8b3ed49e..bb7e732e08 100644 --- a/kubernetes/apps/security/crowdsec/bouncers/envoy/referencegrant.yaml +++ b/kubernetes/apps/security/crowdsec/bouncers/envoy/referencegrant.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/referencegrant_v1beta1.json apiVersion: gateway.networking.k8s.io/v1beta1 kind: ReferenceGrant metadata: diff --git a/kubernetes/apps/web3/monero/xmrig/priorityclass.yaml b/kubernetes/apps/web3/monero/xmrig/priorityclass.yaml index 4592d993b5..f0d587daeb 100644 --- a/kubernetes/apps/web3/monero/xmrig/priorityclass.yaml +++ b/kubernetes/apps/web3/monero/xmrig/priorityclass.yaml @@ -1,4 +1,5 @@ --- +# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/scheduling.k8s.io/priorityclass_v1.json apiVersion: scheduling.k8s.io/v1 kind: PriorityClass metadata: From a2bcf9d40d3c4f8f9674485566882e6d30fb39ff Mon Sep 17 00:00:00 2001 From: Tanguille <91473554+Tanguille@users.noreply.github.com> Date: Mon, 27 Jul 2026 18:12:50 +0200 Subject: [PATCH 2/8] chore(toolhive): drop the five no-op MCPToolConfigs Five MCPToolConfigs carried `toolsFilter: []`, which the CRD documents as "If empty, all tools are exposed" - they filtered nothing, and the ten toolConfigRef fields pointing at them were dead weight. Verified on the live cluster rather than from the doc text, because the two paths could have diverged in the operator. Removed the toolConfigRef from MCPServer kubesearch with `kubectl patch` and re-ran an MCP tools/list against the vmcp-resources gateway: 23 tools before, 23 after, names byte-identical, including all ten kubesearch_* entries. Live state has been restored. That test also retired the reason this was previously skipped. The concern was that restarting the MCP servers would park the gateways for several minutes; the observed recovery was 34 seconds, and the aggregating gateway pods never restarted at all. NOT removed, because these three do real work despite looking similar: - grafana: a genuine 33-tool allow list. - karakeep and searxng: no toolsFilter at all - they use toolsOverride to rename tools (search-bookmarks -> search_bookmarks). Dropping those refs would have renamed eleven live tools. The orphaned `github` MCPToolConfig is left alone here; that file is rewritten on feat/github-mcp-http. --- .../apps/ai/toolhive/config/flux-operator.yaml | 14 +------------- .../apps/ai/toolhive/config/homeassistant.yaml | 14 +------------- kubernetes/apps/ai/toolhive/config/kubesearch.yaml | 14 +------------- .../apps/ai/toolhive/config/postgres-mcp.yaml | 14 +------------- kubernetes/apps/ai/toolhive/config/talos-mcp.yaml | 14 +------------- 5 files changed, 5 insertions(+), 65 deletions(-) diff --git a/kubernetes/apps/ai/toolhive/config/flux-operator.yaml b/kubernetes/apps/ai/toolhive/config/flux-operator.yaml index 0375e8d7a4..f0d765f30f 100644 --- a/kubernetes/apps/ai/toolhive/config/flux-operator.yaml +++ b/kubernetes/apps/ai/toolhive/config/flux-operator.yaml @@ -3,7 +3,7 @@ apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: - name: &name flux-operator + name: flux-operator spec: image: ghcr.io/controlplaneio-fluxcd/flux-operator-mcp:v0.56.0 transport: stdio @@ -14,8 +14,6 @@ spec: env: - name: KUBECONFIG value: /kubeconfig/kubeconfig - toolConfigRef: - name: *name podTemplateSpec: spec: volumes: @@ -61,8 +59,6 @@ spec: env: - name: KUBECONFIG value: /kubeconfig/kubeconfig - toolConfigRef: - name: flux-operator podTemplateSpec: spec: volumes: @@ -92,11 +88,3 @@ spec: limits: cpu: 500m memory: 200Mi ---- -# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcptoolconfig_v1beta1.json -apiVersion: toolhive.stacklok.dev/v1beta1 -kind: MCPToolConfig -metadata: - name: flux-operator -spec: - toolsFilter: [] diff --git a/kubernetes/apps/ai/toolhive/config/homeassistant.yaml b/kubernetes/apps/ai/toolhive/config/homeassistant.yaml index 6f5dc0e4eb..23f87ce951 100644 --- a/kubernetes/apps/ai/toolhive/config/homeassistant.yaml +++ b/kubernetes/apps/ai/toolhive/config/homeassistant.yaml @@ -3,7 +3,7 @@ apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: - name: &name homeassistant + name: homeassistant spec: image: ghcr.io/homeassistant-ai/ha-mcp:7.14.2@sha256:7917b2d385e16e43f45f92fc72a757e5c0aec8d88b3cd69fe64f3b5106cbfe36 transport: streamable-http @@ -22,8 +22,6 @@ spec: - name: toolhive-secrets key: HOMEASSISTANT_API_TOKEN targetEnvName: HOMEASSISTANT_TOKEN - toolConfigRef: - name: *name podTemplateSpec: spec: volumes: @@ -86,8 +84,6 @@ spec: - name: toolhive-secrets key: HOMEASSISTANT_API_TOKEN targetEnvName: HOMEASSISTANT_TOKEN - toolConfigRef: - name: homeassistant podTemplateSpec: spec: volumes: @@ -127,11 +123,3 @@ spec: limits: cpu: 200m memory: 128Mi ---- -# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcptoolconfig_v1beta1.json -apiVersion: toolhive.stacklok.dev/v1beta1 -kind: MCPToolConfig -metadata: - name: homeassistant -spec: - toolsFilter: [] diff --git a/kubernetes/apps/ai/toolhive/config/kubesearch.yaml b/kubernetes/apps/ai/toolhive/config/kubesearch.yaml index fa20c8b402..bdd2be54a2 100644 --- a/kubernetes/apps/ai/toolhive/config/kubesearch.yaml +++ b/kubernetes/apps/ai/toolhive/config/kubesearch.yaml @@ -3,7 +3,7 @@ apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: - name: &name kubesearch + name: kubesearch spec: image: ghcr.io/perfectra1n/kubesearch-mcp:master transport: streamable-http @@ -13,8 +13,6 @@ spec: env: - name: KUBESEARCH_REFRESH_HOURS value: "24" - toolConfigRef: - name: *name podTemplateSpec: spec: volumes: @@ -70,8 +68,6 @@ spec: env: - name: KUBESEARCH_REFRESH_HOURS value: "24" - toolConfigRef: - name: kubesearch podTemplateSpec: spec: volumes: @@ -112,11 +108,3 @@ spec: limits: cpu: 200m memory: 256Mi ---- -# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcptoolconfig_v1beta1.json -apiVersion: toolhive.stacklok.dev/v1beta1 -kind: MCPToolConfig -metadata: - name: kubesearch -spec: - toolsFilter: [] diff --git a/kubernetes/apps/ai/toolhive/config/postgres-mcp.yaml b/kubernetes/apps/ai/toolhive/config/postgres-mcp.yaml index c1dccde55c..307880f05b 100644 --- a/kubernetes/apps/ai/toolhive/config/postgres-mcp.yaml +++ b/kubernetes/apps/ai/toolhive/config/postgres-mcp.yaml @@ -3,7 +3,7 @@ apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: - name: &name postgres-mcp + name: postgres-mcp spec: image: crystaldba/postgres-mcp:0.3.0 transport: stdio @@ -15,8 +15,6 @@ spec: - name: toolhive-secrets key: POSTGRES_MCP_DATABASE_URI targetEnvName: DATABASE_URI - toolConfigRef: - name: *name podTemplateSpec: spec: containers: @@ -52,8 +50,6 @@ spec: - name: toolhive-secrets key: POSTGRES_MCP_DATABASE_URI targetEnvName: DATABASE_URI - toolConfigRef: - name: postgres-mcp podTemplateSpec: spec: containers: @@ -72,11 +68,3 @@ spec: limits: cpu: 500m memory: 512Mi ---- -# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcptoolconfig_v1beta1.json -apiVersion: toolhive.stacklok.dev/v1beta1 -kind: MCPToolConfig -metadata: - name: postgres-mcp -spec: - toolsFilter: [] diff --git a/kubernetes/apps/ai/toolhive/config/talos-mcp.yaml b/kubernetes/apps/ai/toolhive/config/talos-mcp.yaml index 816198f392..4606200742 100644 --- a/kubernetes/apps/ai/toolhive/config/talos-mcp.yaml +++ b/kubernetes/apps/ai/toolhive/config/talos-mcp.yaml @@ -11,7 +11,7 @@ spec: apiVersion: toolhive.stacklok.dev/v1beta1 kind: MCPServer metadata: - name: &name talos-mcp + name: talos-mcp spec: image: registry.erwanleboucher.dev/eleboucher/talos-mcp:0.0.2 transport: streamable-http @@ -31,8 +31,6 @@ spec: value: http - name: TALOS_MCP_HTTP_ADDR value: :8080 - toolConfigRef: - name: *name podTemplateSpec: spec: volumes: @@ -85,8 +83,6 @@ spec: value: http - name: TALOS_MCP_HTTP_ADDR value: :8080 - toolConfigRef: - name: talos-mcp podTemplateSpec: spec: volumes: @@ -114,11 +110,3 @@ spec: limits: cpu: 200m memory: 128Mi ---- -# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/toolhive.stacklok.dev/mcptoolconfig_v1beta1.json -apiVersion: toolhive.stacklok.dev/v1beta1 -kind: MCPToolConfig -metadata: - name: talos-mcp -spec: - toolsFilter: [] From fbf967f361550f4109af30b7bdef103205ab340e Mon Sep 17 00:00:00 2001 From: Tanguille <91473554+Tanguille@users.noreply.github.com> Date: Mon, 27 Jul 2026 18:27:18 +0200 Subject: [PATCH 3/8] docs(llm-hosting): make the benchmark record generic and keep the open experiments Renamed sglang-benchmarks.md to engine-benchmarks-gfx1201.md: the file was never SGLang-specific - it is a vLLM / SGLang / llama.cpp comparison for Qwen3.6-27B on RDNA4, and its 2026-06-21 round actually concluded in favour of vLLM. Restored the "Open / next experiments" list, which the previous commit wrongly deleted as stale scaffolding. Five of those are outstanding work, not history. Added the SGLang revalidation item: when upstream rebases the RDNA4 patch series past our FORK_REF pin, re-run the bench/ concurrency sweep and verbatim test and compare against the recorded v0.5.15 numbers before moving the production pin. Reframed the header from "SUPERSEDED / historical" to a dated measurement series with a live experiment queue, and dropped only the generic Process Instructions boilerplate. Updated the one inbound link in sglang-blockers.md. --- ...hmarks.md => engine-benchmarks-gfx1201.md} | 46 +++++++++++++++++-- docs/llm-hosting/sglang-blockers.md | 2 +- 2 files changed, 43 insertions(+), 5 deletions(-) rename docs/llm-hosting/{sglang-benchmarks.md => engine-benchmarks-gfx1201.md} (91%) diff --git a/docs/llm-hosting/sglang-benchmarks.md b/docs/llm-hosting/engine-benchmarks-gfx1201.md similarity index 91% rename from docs/llm-hosting/sglang-benchmarks.md rename to docs/llm-hosting/engine-benchmarks-gfx1201.md index c98eb01ce9..0fa2eda395 100644 --- a/docs/llm-hosting/sglang-benchmarks.md +++ b/docs/llm-hosting/engine-benchmarks-gfx1201.md @@ -1,8 +1,14 @@ -# Qwen3.6-27B serving on R9700 (gfx1201) — Benchmarks & Decision +# Serving-engine benchmarks on R9700 (gfx1201) -**SUPERSEDED:** production runs SGLang (baked `ghcr.io/tanguille/sglang-rdna4` image) — see -`docs/llm-hosting/sglang-blockers.md` and `docker/sglang-rdna4/README.md` for the current -engine decision. This doc is the historical vLLM-vs-SGLang record from 2026-06-21. +Multi-engine measurement record for Qwen3.6-27B on RDNA4 — vLLM, SGLang and llama.cpp. +Measurements are dated and kept as a historical series; the open experiments at the +bottom are still outstanding. + +**Current production engine: SGLang**, baked `ghcr.io/tanguille/sglang-rdna4`. Build, +pin and rollback mechanics are in `docker/sglang-rdna4/README.md`; outstanding upstream +gaps are tracked in `docs/llm-hosting/sglang-blockers.md`. The 2026-06-21 round below +concluded in favour of vLLM and has since been superseded — read its numbers as a +snapshot of that date, not as current guidance. ## Round 2 Final Results — 2026-06-21 @@ -493,3 +499,35 @@ verify). The max concurrency sweet spot for QoS is C=8. **Note:** vLLM auto-set `max_num_scheduled_tokens=2048` based on spec config. This limits token budget per scheduling step. No impact observed at C≤16 with out=200. + +--- + +## Open / next experiments + +Done, kept for the result: + +- [x] **bf16 KV cache** — *slower* (12.35 vs 14.96) and half capacity. fp8 KV wins on sglang + (native gfx1201 fp8 kernels). Keep fp8_e4m3. +- [x] **vLLM kyuz0 batching** — 48.6 tok/s @ C8, 85.8 tok/s @ C16, APC on; CUDAGraphs work on + gfx1201 in 0.22.1rc1. +- [x] **vLLM FP8 KV + APC coexistence** — 128.9 tok/s @ C32, 50% over the FP16 baseline; bug + #13147 does not affect kyuz0. +- [x] **VLLM_ROCM_USE_AITER=1** — fails: GDN kernel autotune exhausts all configs <= 64 KiB + SRAM on gfx1201. AITER config tables target MI300X. +- [x] **MTP speculative decoding** — `--spec-method mtp --spec-tokens 4`. C=1 18.4 tok/s (2.6x), + C=8 126.8 tok/s aggregate (2.7x). Qwen3.6-27B ships MTP heads, no draft model needed. + +Still outstanding: + +- [ ] **Revalidate the latest SGLang** once upstream rebases the RDNA4 patch series past our + `FORK_REF` pin. Blocked on the fork, not on us — see blocker 7 in `sglang-blockers.md`. + Re-run the concurrency sweep and the verbatim-reproduction test from `bench/` and compare + against the recorded v0.5.15 numbers before moving the production pin. +- [ ] vLLM fp8 KV + extended context: `--max-model-len 131072` with fp8 KV — does 32K -> 131K + move throughput? +- [ ] Push MTP concurrency beyond 16 with `max_num_seqs=32` — does C=32 MTP reach 200+ tok/s? +- [ ] `mamba_track_interval` / scheduler tuning under Config B for higher concurrency. +- [ ] Quality probe (logprob/eval) to close out the fork's flagged torch-2.12 attention drift; + basic coherence already passes. +- [ ] llama.cpp gfx1201 tuning sweep — `-b`/`-ub`, hipBLASLt env, perf level, iGPU/`-sm`, Vulkan + backend. Then cache probe and concurrency sweep. diff --git a/docs/llm-hosting/sglang-blockers.md b/docs/llm-hosting/sglang-blockers.md index 286c609dc1..676e976801 100644 --- a/docs/llm-hosting/sglang-blockers.md +++ b/docs/llm-hosting/sglang-blockers.md @@ -266,7 +266,7 @@ not the kernel, that makes overlap a no-op). **Context:** AMD Quark 0.12.0 (2026-07) adds AMDFP4 (E5M3 per-block scales), NVFP4, native MXFP4 inference support, and the SVDQuant algorithm (INT4/MXFP4/NVFP4 weights + low-rank outlier branch). Tempting as a path off the slow AWQ-int4 dense-decode wall. **Root cause (why none of it helps this box):** Quark is a *quantization producer* — it emits checkpoints. Our dense-decode bottleneck is the RDNA4 **inference kernel** (Triton W4A16 + GatedDeltaNet), an SGLang/fork problem the checkpoint format can't touch. Re-quantizing the same weights to a different FP4 format doesn't change which kernel SGLang dispatches at decode. Producer with no consumer: -- **MXFP4 / NVFP4 / AMDFP4:** no fused decode kernel in the mattbucci fork for gfx1201 (consistent with the MXFP4/fp4 "ruled out for dense" finding in `sglang-benchmarks.md`). +- **MXFP4 / NVFP4 / AMDFP4:** no fused decode kernel in the mattbucci fork for gfx1201 (consistent with the MXFP4/fp4 "ruled out for dense" finding in `engine-benchmarks-gfx1201.md`). - **SVDQuant:** its runtime is **Nunchaku — CUDA W4A4 + low-rank, Nvidia-only**; no ROCm path, and SGLang has no consumer for the SVDQuant (INT4 + low-rank branch) format on any GPU. Public SVDQuant checkpoints are also almost all diffusion/image models (FLUX, Qwen-Image); no LLM checkpoint exists for Qwen3.6-27B (even Qwen3.5-27B is GPTQ-Int4/AWQ only). **Investigated:** 2026-07-04, doc-only (no build). SVDQuant quality-vs-AWQ was the one non-throughput angle (better int4 accuracy at equal speed) but we've flagged no quality issue, so not worth the self-quantize + missing-kernel effort. From 0dcb2b95598b9aece74ae465361ee1961cbdfc05 Mon Sep 17 00:00:00 2001 From: Tanguille <91473554+Tanguille@users.noreply.github.com> Date: Mon, 27 Jul 2026 18:31:46 +0200 Subject: [PATCH 4/8] chore(llmkube): drop the redundant inProgress health-check expression Flux's CEL status evaluator returns InProgressStatus when no expression matches (fluxcd/pkg runtime/cel/status_evaluator.go, final return), so an inProgress expression only earns its place if its truth set overlaps failed or current. Here it cannot. The InferenceService phase enum is Pending, Creating, Progressing, Ready, WaitingForGPU, Stopped, Suspended, Failed. inProgress matched Pending/Creating/Progressing/WaitingForGPU, failed matches Failed, current matches Stopped or Ready+Available - three disjoint sets over one scalar field, so every phase lands on the same verdict with the expression removed. Suspended already fell through to InProgress and still does. Also brings this in line with the only two other healthCheckExprs in the repo (cert-manager, rook-ceph), which both use failed + current alone. --- kubernetes/apps/ai/llmkube/ks.yaml | 1 - 1 file changed, 1 deletion(-) diff --git a/kubernetes/apps/ai/llmkube/ks.yaml b/kubernetes/apps/ai/llmkube/ks.yaml index e7de9bed08..01a97000d3 100644 --- a/kubernetes/apps/ai/llmkube/ks.yaml +++ b/kubernetes/apps/ai/llmkube/ks.yaml @@ -45,7 +45,6 @@ spec: healthCheckExprs: - apiVersion: inference.llmkube.dev/v1alpha1 kind: InferenceService - inProgress: has(status.phase) && status.phase in ['Pending', 'Creating', 'Progressing', 'WaitingForGPU'] failed: has(status.phase) && status.phase == 'Failed' # Stopped is terminal; the CRD says treat it as Pending, which deadlocks this gate. current: has(status.phase) && (status.phase == 'Stopped' || (status.phase == 'Ready' && has(status.conditions) && status.conditions.exists(e, e.type == 'Available' && e.status == 'True'))) From cab9b5c56930e452d1e545eac0ad7784aa9a6c01 Mon Sep 17 00:00:00 2001 From: Tanguille <91473554+Tanguille@users.noreply.github.com> Date: Mon, 27 Jul 2026 20:11:19 +0200 Subject: [PATCH 5/8] docs(sglang): make the emergency rebuild path actionable again Review caught that sglang-blockers.md and docker/sglang-rdna4/README.md disagreed about the retired PVC-rebuild script. The README also told readers to keep the Dockerfile in sync with a file that no longer exists, and linked the cutover doc this branch removes. Both now name the exact recovery command instead of gesturing at git history: `git show b8f12ae75^:kubernetes/apps/ai/sglang/app/scripts/sglang-env-rebuild.sh`, which was checked and returns all 151 lines. The "keep them in sync" instruction is gone, since the Dockerfile is now the only build path. --- docker/sglang-rdna4/README.md | 15 +++++++++------ 1 file changed, 9 insertions(+), 6 deletions(-) diff --git a/docker/sglang-rdna4/README.md b/docker/sglang-rdna4/README.md index 1a902cb2ed..126932f464 100644 --- a/docker/sglang-rdna4/README.md +++ b/docker/sglang-rdna4/README.md @@ -2,18 +2,21 @@ Serves **Qwen3.6-27B int4-AWQ** on a single **AMD Radeon AI PRO R9700** (RDNA4 / gfx1201) via SGLang + the RDNA4 patch fork. This is the **git-reproducible replacement** for the -old runtime-from-PVC env (`scripts/sglang-env-rebuild.sh`); see -[`docs/llm-hosting/sglang-oci-cutover.md`](../../docs/llm-hosting/sglang-oci-cutover.md) -for the why and the cutover. +old runtime-from-PVC env, and is now the consolidated reference for building, pinning and +rolling back the image. The cutover plan it replaced was removed once complete; recover it +with `git log --diff-filter=D -- docs/llm-hosting/sglang-oci-cutover.md`. The image vendors [`mattbucci/2x-R9700-RDNA4-GFX1201-sglang-inference`](https://github.com/mattbucci/2x-R9700-RDNA4-GFX1201-sglang-inference) at a pinned commit, runs its own `scripts/setup.sh` (the only build path the fork supports — conda-only, no upstream Dockerfile), then applies the three TP=1 fixes `setup.sh` omits. **Re-pinning `FORK_REF` (and `SGLANG_TAG` on a version bump) is the whole maintenance surface.** -> The `Dockerfile` mirrors `kubernetes/apps/ai/sglang/app/scripts/sglang-env-rebuild.sh` -> exactly (same `FORK_REF`, `SGLANG_TAG`, and the three patches). Keep them in sync — that -> script remains the emergency PVC-rebuild fallback if the registry is ever unreachable. +> The `Dockerfile` was derived from the retired `sglang-env-rebuild.sh` (same `FORK_REF`, +> `SGLANG_TAG`, and the three patches). That script went with the `kubernetes/apps/ai/sglang` +> directory in b8f12ae75; if the registry is ever unreachable and the PVC-rebuild path is +> needed, recover it with +> `git show b8f12ae75^:kubernetes/apps/ai/sglang/app/scripts/sglang-env-rebuild.sh`. +> There is no longer a second copy to keep in sync — this Dockerfile is the only build path. ## Pins (verified against `env-rebuild.sh`, 2026-07-12) From 35b3a2f5dac7a25164cc000c23ade5072979d98b Mon Sep 17 00:00:00 2001 From: Tanguille <91473554+Tanguille@users.noreply.github.com> Date: Mon, 27 Jul 2026 20:41:08 +0200 Subject: [PATCH 6/8] feat(sglang): enable HiCache L3 file backend L3 was recorded as failing restart recovery on 07-15. Two defects sat upstream of that measurement: the hash extension could not JIT-compile without OpenSSL headers in the runtime stage, and write_through_selective gated promotion on hit_count >= 2 so first-pass content never reached L2 for L3 to back up. Retested 07-27 on sha256:72d934d3 under plain write_through. After a cold restart, prefetched_tokens_total{storage_backend="file"} read 63,205, and L2 accounting closes exactly: 73,120 newly backed + 63,205 prefetched = 136,325 hicache_host_used_tokens. L3 gets its own PVC rather than sharing the triton cache, and the evictor is bounded: it defaults to unbounded, and openebs-hostpath enforces no quota. MIN_FREE_SPACE guards the node's shared root independently of cache size. --- docs/llm-hosting/sglang-blockers.md | 38 +++++++++++++++---- docs/llm-hosting/vllm-vs-sglang-2026-07.md | 18 +++++++-- .../ai/llmkube/models/qwen36-27b-sglang.yaml | 36 +++++++++++++++++- 3 files changed, 78 insertions(+), 14 deletions(-) diff --git a/docs/llm-hosting/sglang-blockers.md b/docs/llm-hosting/sglang-blockers.md index 676e976801..ab3afbcb1c 100644 --- a/docs/llm-hosting/sglang-blockers.md +++ b/docs/llm-hosting/sglang-blockers.md @@ -195,10 +195,9 @@ the pinned fork tree carries stock v0.5.14 hicache code (no fork patch touches i population request took 113.3s. After a clean pod restart, the token-exact post-restart extension took 70.7s, but the file-backend metric recorded only 4,538 backed-up tokens, with zero storage prefetches and zero `storage_HiCacheFile` hits. Aggregate completion throughput for C1/C4/C8 was -11.16/8.74/14.09 tok/s; the C8 sample overlapped real traffic. The validated direct I/O, ratio 1.5 -L2, and `write_through_selective` policy remain enabled; do not propose unverified cache tweaks. -Rollback does not remove `/cache/sglang/hicache`; it can contain prompt-derived data and requires -explicit operator approval before deletion. +11.16/8.74/14.09 tok/s; the C8 sample overlapped real traffic. Direct I/O and ratio 1.5 L2 remain +validated and enabled; `write_through_selective` was replaced on 07-27, see the retest below. Do +not propose unverified cache tweaks. **Correction (2026-07-27): "failed restart recovery" was the wrong conclusion.** Two defects sat upstream of that measurement, and neither is a property of L3. @@ -220,10 +219,33 @@ upstream of that measurement, and neither is a property of L3. 4,538-of-61,600 backup figure. Any L3 test must run under plain `write_through`. The 07-15 run did not crash, so it cannot have been on an image with defect 1 — meaning it is not -comparable to current builds and should not be quoted as evidence either way. **L3 is untested, not -failed.** Retest sequence: rebuild with `libssl-dev` → `write_through` → synthetic prefill to prove -the backend initialises → populate/restart/extension probe. Not on the production replica: the -07-27 attempt took prod down mid-traffic and cost ~40 min across two reloads. +comparable to current builds and should not be quoted as evidence either way. **L3 was untested, +not failed.** + +**Retest (2026-07-27): L3 works.** Run on image `sha256:72d934d3` under plain `write_through`, +with `--hicache-storage-backend file` and `SGLANG_HICACHE_FILE_BACKEND_STORAGE_DIR`. + +| stage | evidence | +|---|---| +| backend initialises | synthetic prefill served; pod ran 22 min under live traffic, 0 restarts | +| L3 writes | 268,176 files / 14 GB on disk, `backuped_tokens_total` 267,978 | +| L2 saturates | `hicache_host_used_tokens` 267,978 of 274,861 | +| survives restart | after a cold start, `prefetched_tokens_total{storage_backend="file"}` 63,205 | + +The restart figures close exactly: 73,120 newly backed + 63,205 prefetched = 136,325 +`hicache_host_used_tokens`, so every L2 token is accounted for as either a fresh write or a +read-back from disk. Populate (58,071 tok) PP 196.9 tok/s; extension probe (58,209 tok) PP +252.9 tok/s, TG 0.56 tok/s. Both ran with production traffic on the GPU, so both are floors, +not clean numbers. + +Two operational notes. The evictor is unbounded by default and `openebs-hostpath` enforces no +quota, so `MAX_SIZE` (64Gi) and `MIN_FREE_SPACE` (100Gi) are the only things between L3 and +control-1's shared 500G root; the free-space floor is what keeps a full cache from pushing the +node under kubelet's ~50G nodefs eviction threshold. L3 lives on its own `qwen36-27b-hicache` +PVC at `/hicache` rather than sharing the triton cache, so its footprint is visible where +capacity is planned. Rollback does not remove that directory; it holds prompt-derived data and +needs explicit operator approval before deletion. The 07-27 test data still sits at the old +`/cache/sglang/hicache` path and is orphaned by this move. **`hicache-write-policy write_back` trialled and reverted (2026-07-09):** kept alongside the L3 removal above as a still-valid L1→L2 (GPU→host) optimization — synthetic testing showed 0 aborts and lower diff --git a/docs/llm-hosting/vllm-vs-sglang-2026-07.md b/docs/llm-hosting/vllm-vs-sglang-2026-07.md index b9d9bbfc89..6a9b1ec22c 100644 --- a/docs/llm-hosting/vllm-vs-sglang-2026-07.md +++ b/docs/llm-hosting/vllm-vs-sglang-2026-07.md @@ -129,8 +129,9 @@ Three things combine: impossible. 2. **ETag is not integrity.** The etag file is written from response headers, which arrive before the body, so it can persist for a transfer that later dies. The next - boot sends `If-None-Match`, gets `304`, and prints `Model artifact ... revalidated` - over a corrupt file. Nothing checks size against `Content-Length`. + boot sends `If-None-Match` and prints `Model artifact ... revalidated` over a corrupt + file. Nothing checks size against `Content-Length`. Note the message is printed on any + curl exit 0, not on a `304`, so it carries no information about what actually happened. 3. **The fallback keeps bad data.** `elif [ -f "$dest" ]` treats presence as validity. There is currently no content-integrity mechanism at all: sha256 pinning was rejected on @@ -175,9 +176,18 @@ served over a truncated weight file. For LFS artifacts that cannot happen, since always re-fetched. The stale-etag trap is real for config files and for the *bookkeeping*, but the practical failure mode on weights is cost, not silent corruption. +A third defect compounds both: `-o "$dest"` opens the destination for writing at request +time, so the valid cached copy is truncated at the start of every restart before curl knows +whether it needs the body. The cache is not merely ignored, it is destroyed. + +Confirmed again on the 07-27 L3 restart, which is the clearest single reading: all nine +artifacts logged `revalidated` while `model-downloader` ran 14m21s (18:08:38Z to 18:22:59Z), +which at the observed rate is the whole weight set. + This is operator-side (the downloader script is generated by llmkube), so it needs an -upstream fix: follow redirects with the conditional intact, or range/size check the local -file before deciding to transfer. +upstream fix: `HEAD` and compare `Content-Length` before transferring, download to +`$dest.tmp` + `mv`, and log `revalidated` only on an actual `304`. Filed as +[defilantech/LLMKube#1309](https://github.com/defilantech/LLMKube/issues/1309). ## Open items diff --git a/kubernetes/apps/ai/llmkube/models/qwen36-27b-sglang.yaml b/kubernetes/apps/ai/llmkube/models/qwen36-27b-sglang.yaml index a4bfe1d5a5..1c9673e8c5 100644 --- a/kubernetes/apps/ai/llmkube/models/qwen36-27b-sglang.yaml +++ b/kubernetes/apps/ai/llmkube/models/qwen36-27b-sglang.yaml @@ -25,6 +25,20 @@ spec: requests: storage: 10Gi --- +apiVersion: v1 +kind: PersistentVolumeClaim +metadata: + name: qwen36-27b-hicache + namespace: ai +spec: + accessModes: + - ReadWriteOnce + storageClassName: openebs-hostpath + resources: + # hostpath enforces no quota, so the real bound is the evictor's MAX_SIZE below. + requests: + storage: 64Gi +--- # yaml-language-server: $schema=https://k8s-schemas.home-operations.com/inference.llmkube.dev/model_v1alpha1.json apiVersion: inference.llmkube.dev/v1alpha1 kind: Model @@ -63,7 +77,8 @@ spec: runtime: sglang # Operator default binds :: (IPv6-only on this IPv4 cluster); override to v4. bindAddress: 0.0.0.0 - image: ghcr.io/tanguille/sglang-rdna4:v0.5.15-gfx1201@sha256:c0372d10bdc4baebd50a9726896e214681d703e1ce6c9032bbef0cc4f82bca76 + # L3 needs OpenSSL headers in the runtime stage; earlier digests kill the scheduler. + image: ghcr.io/tanguille/sglang-rdna4:v0.5.15-gfx1201@sha256:72d934d335529934bf73cd480b728cd08ae61178c9204dc37a025302b31fd61e containerPort: 30000 endpoint: # v0.9.8 maps endpoint.port to both Service port and targetPort; use the @@ -107,10 +122,22 @@ spec: value: "0" - name: TRITON_CACHE_DIR value: /cache/sglang/triton + - name: SGLANG_HICACHE_FILE_BACKEND_STORAGE_DIR + value: /hicache + # Unset defaults to unbounded; a full L2 measured 14G on 2026-07-27. + - name: SGLANG_HICACHE_FILE_BACKEND_MAX_SIZE + value: 64Gi + # MAX_SIZE alone is blind to the other tenants of control-1's shared 500G root, so + # hold a floor well above kubelet's ~50G nodefs eviction threshold. + - name: SGLANG_HICACHE_FILE_BACKEND_MIN_FREE_SPACE + value: 100Gi extraVolumes: - name: triton-cache persistentVolumeClaim: claimName: qwen36-27b-triton-cache + - name: hicache + persistentVolumeClaim: + claimName: qwen36-27b-hicache - name: dshm emptyDir: medium: Memory @@ -118,6 +145,8 @@ spec: extraVolumeMounts: - name: triton-cache mountPath: /cache + - name: hicache + mountPath: /hicache - name: dshm mountPath: /dev/shm probeOverrides: @@ -181,5 +210,8 @@ spec: - --schedule-conservativeness - "0.1" - --enable-mixed-chunk + # write_through_selective gates promotion on hit_count >= 2, which left L3 empty. - --hicache-write-policy - - write_through_selective + - write_through + - --hicache-storage-backend + - file From 29343f89703ace05499c4eb2ab747beee463db7f Mon Sep 17 00:00:00 2001 From: Tanguille <91473554+Tanguille@users.noreply.github.com> Date: Mon, 27 Jul 2026 21:20:55 +0200 Subject: [PATCH 7/8] fix(opencode): use the app-template schema, not the generic HelmRelease one opencode renders from app-template, and the repo pairs that chart with the bjw-s schema in 52 of 53 cases. --- kubernetes/apps/ai/opencode/app/helmrelease.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/kubernetes/apps/ai/opencode/app/helmrelease.yaml b/kubernetes/apps/ai/opencode/app/helmrelease.yaml index c026f34e7d..f4de2907e1 100644 --- a/kubernetes/apps/ai/opencode/app/helmrelease.yaml +++ b/kubernetes/apps/ai/opencode/app/helmrelease.yaml @@ -1,5 +1,5 @@ --- -# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/helm.toolkit.fluxcd.io/helmrelease_v2.json +# yaml-language-server: $schema=https://raw.githubusercontent.com/bjw-s-labs/helm-charts/main/charts/other/app-template/schemas/helmrelease-helm-v2.schema.json apiVersion: helm.toolkit.fluxcd.io/v2 kind: HelmRelease metadata: From 70b5fae4fd8d4b7ba187dd0e8965ba528b7a8a7b Mon Sep 17 00:00:00 2001 From: Tanguille <91473554+Tanguille@users.noreply.github.com> Date: Mon, 27 Jul 2026 21:32:21 +0200 Subject: [PATCH 8/8] fix(sglang): namespace the HiCache dir by engine and weights revision The page filename is sha256(token_ids, prior_hash, page_size) plus _{model_name}_{tp_rank}_{tp_size}, so it encodes neither the weights revision nor the KV layout. Renovate bumps both the AWQ revision and the sglang image, and either would have served KV pages computed by the previous one. Also drop the claim that a Content-Length compare validates the download cache. --- docs/llm-hosting/vllm-vs-sglang-2026-07.md | 7 +++++-- kubernetes/apps/ai/llmkube/models/qwen36-27b-sglang.yaml | 8 +++++++- 2 files changed, 12 insertions(+), 3 deletions(-) diff --git a/docs/llm-hosting/vllm-vs-sglang-2026-07.md b/docs/llm-hosting/vllm-vs-sglang-2026-07.md index 6a9b1ec22c..f298b810c0 100644 --- a/docs/llm-hosting/vllm-vs-sglang-2026-07.md +++ b/docs/llm-hosting/vllm-vs-sglang-2026-07.md @@ -185,8 +185,11 @@ artifacts logged `revalidated` while `model-downloader` ran 14m21s (18:08:38Z to which at the observed rate is the whole weight set. This is operator-side (the downloader script is generated by llmkube), so it needs an -upstream fix: `HEAD` and compare `Content-Length` before transferring, download to -`$dest.tmp` + `mv`, and log `revalidated` only on an actual `304`. Filed as +upstream fix: download to `$dest.tmp` + `mv` so a fetch can never truncate the copy it is +revalidating, and log `revalidated` only on an actual `304`. A `HEAD` `Content-Length` +compare is worth adding as a cheap truncation preflight, but it is not cache validation: +equal size proves neither freshness nor integrity, so it supplements the conditional GET +rather than replacing it. Filed as [defilantech/LLMKube#1309](https://github.com/defilantech/LLMKube/issues/1309). ## Open items diff --git a/kubernetes/apps/ai/llmkube/models/qwen36-27b-sglang.yaml b/kubernetes/apps/ai/llmkube/models/qwen36-27b-sglang.yaml index 1c9673e8c5..0a63b0fb75 100644 --- a/kubernetes/apps/ai/llmkube/models/qwen36-27b-sglang.yaml +++ b/kubernetes/apps/ai/llmkube/models/qwen36-27b-sglang.yaml @@ -122,8 +122,14 @@ spec: value: "0" - name: TRITON_CACHE_DIR value: /cache/sglang/triton + # Correctness knob, not a layout preference. The on-disk page name is + # sha256(token_ids, prior_hash, page_size) + "_{model_name}_{tp_rank}_{tp_size}", + # so nothing in it encodes the weights revision, the KV dtype or the KV layout. + # Same served-model-name plus same tokens hits the same file, which means a + # Renovate bump of either the AWQ revision below or the sglang image would serve + # KV pages computed by the previous one. Bump this suffix with either of them. - name: SGLANG_HICACHE_FILE_BACKEND_STORAGE_DIR - value: /hicache + value: /hicache/sglang-v0.5.15_awq-f541031d # Unset defaults to unbounded; a full L2 measured 14G on 2026-07-27. - name: SGLANG_HICACHE_FILE_BACKEND_MAX_SIZE value: 64Gi