From c1b3cbfbdef028a5fe0f45c9cf77bc21df2d039f Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Tue, 11 Aug 2026 13:17:19 +0000 Subject: [PATCH 01/25] Studio: drop the speculative drafter under Auto when only the model fits in VRAM --- studio/backend/core/inference/llama_cpp.py | 119 ++++++++++++++++- studio/backend/models/inference.py | 3 + .../backend/tests/test_llama_cpp_placement.py | 120 ++++++++++++++++++ .../src/features/chat/chat-settings-sheet.tsx | 2 + .../frontend/src/features/chat/types/api.ts | 3 + 5 files changed, 245 insertions(+), 2 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 0dee67c54e..c07a5e90ca 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -10051,6 +10051,9 @@ def _ubatch_for_slots(slots: int) -> Optional[int]: _vulkan_explicit_unmatched = False _vulkan_requested_ids: list[int] = [] _vulkan_available_ordinals: list[int] = [] + # Auto dropped the drafter because only the target fit. Bound before + # the try for the same reason: the launch below reads it either way. + _spec_dropped_no_vram = False try: gguf_size = self._get_gguf_size_bytes(model_path) # Include GPU-loaded mmproj in the fit budget (#5825). @@ -10456,6 +10459,104 @@ def _cc_split_extra(ctx: int) -> int: # llama.cpp split by free VRAM). tp_tensor_split: Optional[list[int]] = None explicit_ctx = requested_ctx > 0 + _draft_cpu_no_embedded = _draft_on_cpu and not self._nextn_predict_layers + + # When the target pins on GPU but the drafter's reserve is what tips + # it over, Auto drops the drafter: it only buys speed, and paying + # for it with a smaller context (or a --fit offload, where decode + # collapses ~3x) is the worse trade. An explicit dropdown / CLI / + # extras choice is honored and keeps the shrink behaviour below. + # Priced over the WHOLE GPU set, the ceiling for any subset the fit + # then picks, at the context the target alone would get: so it drops + # only when no placement could have held both. Carried to the launch + # as drafter_no_vram, which downgrades the Auto emit to ngram-mod. + # The gate reads the user's choice, not _mtp_effective, which by + # here is the kind Auto resolved and can no longer tell the two apart. + if ( + _mtp_will_engage + and (_canonicalize_spec_mode(speculative_type) or "auto") == "auto" + and not _user_mtp_via_extras + and not _user_draft_via_extras + and not _extra_args_set_spec_type(extra_args) + and not _draft_cpu_no_embedded + and gpus + and effective_ctx > 0 + and self._can_estimate_kv() + ): + _probe_gpus = len(gpus) + + def _probe_base(drafter: bool) -> int: + # model_size_fit's terms, before it is built below. + _soft = self._CUDA_CONTEXT_RESERVE_BYTES + if effective_is_vision and mmproj_size > 0: + _soft += int(mmproj_size * (self._MMPROJ_VRAM_SAFETY - 1.0)) + if drafter: + _soft += self._MTP_DRAFT_COMPUTE_BYTES + return ( + model_size + + _compute_buffer_pipeline + + _soft + + max(0, _probe_gpus - 1) * _pipeline_overhead_bytes + ) + + def _probe_budget(drafter: bool) -> float: + # The flat fraction is the reserve whenever _mtp_bytes is 0. + return _pool_budget_mib( + gpus, + self._GPU_PIN_VRAM_FRACTION + - ( + _MTP_VRAM_RESERVE_FRAC + if (drafter and (mtp_overhead_fn is None or _mtp_kv_unsized)) + else 0.0 + ), + ) + + def _probe_cc(ctx: int) -> int: + return _cc_bytes(ctx, _probe_gpus) + + _base_wo, _budget_wo = _probe_base(False), _probe_budget(False) + # Explicit context is honored verbatim, so that is what the + # drafter has to fit alongside; Auto gets its own best cap. + _ctx_wo = ( + effective_ctx + if explicit_ctx + else self._fit_context_to_vram( + effective_ctx, + _budget_wo, + _base_wo, + cache_type_kv, + swa_full = swa_full, + n_parallel = n_parallel, + kv_unified = planned_kv_unified, + n_ubatch = _effective_ubatch, + flash_attn = planned_flash_attn, + mtp_engaged = False, + mtp_overhead_fn = None, + compute_ctx_bytes_fn = _probe_cc, + budget_frac = 1.0, + total_mib = None, + ) + ) + if _ctx_wo > 0: + _shared = _kv_bytes(_ctx_wo) + _probe_cc(_ctx_wo) + _foot_wo = (_base_wo + _shared) / (1024 * 1024) + _foot_w = ( + _probe_base(True) + _shared + _mtp_bytes(_ctx_wo) + ) / (1024 * 1024) + if _foot_wo <= _budget_wo and _foot_w > _probe_budget(True): + _spec_dropped_no_vram = True + _mtp_will_engage = False + logger.warning( + "Speculative decoding disabled for this load: the model " + "fits in VRAM at context %d but its drafter does not " + "(needs %.1f GB of a %.1f GB budget). Auto keeps the " + "context rather than shrink it for a speed option. " + "Select the drafter in Settings to force it.", + _ctx_wo, + _foot_w / 1024, + _probe_budget(True) / 1024, + ) + # Flat MTP reserve fraction: used only as the fallback when the # byte-accurate mtp_overhead_fn can't size the draft KV (dims # unavailable, or _mtp_kv_unsized = weights-only). A separate @@ -10464,7 +10565,6 @@ def _cc_split_extra(ctx: int) -> int: _flat_mtp_engages = _mtp_will_engage and ( mtp_overhead_fn is None or _mtp_kv_unsized ) - _draft_cpu_no_embedded = _draft_on_cpu and not self._nextn_predict_layers # MTP reserves GPU VRAM unless its only drafter is a separate # CPU-offloaded one (an embedded head stays on GPU). The tensor # path reserves like the layer path; gate both on this. @@ -11428,6 +11528,7 @@ def _restore_after_tensor_downgrade(): mtp_draft_path = (None if _spec_canon == "dspark" else launch_mtp_draft_path), dspark_draft_path = (launch_mtp_draft_path if _spec_canon == "dspark" else None), dspark_fit_sized = not use_fit, + drafter_no_vram = _spec_dropped_no_vram, draft_device = _draft_device, ) # _build_speculative_flags judged the stripped list, so a user @@ -12787,6 +12888,7 @@ def _build_speculative_flags( mtp_draft_path: Optional[str] = None, dspark_draft_path: Optional[str] = None, dspark_fit_sized: bool = True, + drafter_no_vram: bool = False, draft_device: Optional[str] = None, ) -> List[str]: """Return the llama-server flag list for the requested spec mode. @@ -13094,7 +13196,20 @@ def _fallback_drafter_not_found() -> None: # effective_mode == "auto": the promotion path. llama.cpp #22673: # MTP is compatible with mmproj, so there's no vision gate. - if dspark_draft_path and caps.get("supports_dspark"): + if drafter_no_vram: + # The fit found room for the target but not for the drafter's reserve, + # and reserved nothing for it, so emitting one now would OOM the load. + # Same downgrade as the MLA branch below: ngram-mod costs no VRAM, and + # spec-off when the build lacks it. The drafter paths stay recorded, so + # the UI still names the kind and a repeat Apply still dedupes. + self._spec_fallback_reason = "drafter_no_vram" + logger.info( + "Auto: the drafter does not fit in VRAM alongside the model at this " + "context, so it is dropped. Choose it in the Speculative Decoding " + "dropdown to force it at a smaller context." + ) + _emit_ngram_mod() + elif dspark_draft_path and caps.get("supports_dspark"): # DSpark first: load_model only hands a sidecar down once it has one # this binary can launch, and it beats every other Auto outcome for # this architecture (1.84x on 4x B200, 1.91x on one). Without it these diff --git a/studio/backend/models/inference.py b/studio/backend/models/inference.py index 9674437a6c..00f434c1e5 100644 --- a/studio/backend/models/inference.py +++ b/studio/backend/models/inference.py @@ -870,6 +870,9 @@ class InferenceStatusResponse(_InferenceRuntimeFields): "re-enable it (show the update affordance); 'runtime_error' -> the " "current build could not run it; 'drafter_not_found' -> the model's " "separate MTP or DSpark drafter could not be resolved; " + "'drafter_no_vram' -> an Auto-mode fit downgrade: the model pins on " + "GPU but the drafter's reserve does not, and Auto keeps the context " + "rather than shrink it; select the drafter in Settings to force it. " "'mla_mtp_disabled' -> " "an Auto-mode policy downgrade: the model is MLA (GLM-5.2 et al.) " "whose llama.cpp MTP path runs slower than no speculation, so Auto " diff --git a/studio/backend/tests/test_llama_cpp_placement.py b/studio/backend/tests/test_llama_cpp_placement.py index bedfabda2c..d4b74f7c9c 100644 --- a/studio/backend/tests/test_llama_cpp_placement.py +++ b/studio/backend/tests/test_llama_cpp_placement.py @@ -356,3 +356,123 @@ def test_diffusion_does_not_reinterpret_vulkan_ordinals(tmp_path): gpu_ids = [1], ) ) + + +# ── Auto drops a drafter the VRAM cannot hold ───────────────────────── + + +def _tight_vram_backend(tmp_path: Path, *, drafter_gb: float): + """One 24 GB card, a 16 GB target and a drafter of the caller's size. + + The fit terms are stubbed to constants so the only variable is whether the + drafter's reserve clears the pin budget. + """ + gb = 1024**3 + backend, gguf = _backend(tmp_path, vulkan = False, memory = [(0, 24_576, 24_576)]) + sidecar = tmp_path / "dspark-model-Q8_0.gguf" + sidecar.write_bytes(b"draft") + backend._get_gguf_size_bytes = lambda path: ( + int(drafter_gb * gb) if str(path) == str(sidecar) else 16 * gb + ) + backend._can_estimate_kv = lambda: True + backend._estimate_kv_cache_bytes = lambda *args, **kwargs: 1 * gb + backend._compute_buffer_ctx_bytes = lambda *args, **kwargs: 0 + # Positive, or the fit swaps in its 5 GB flat reserve and swamps the numbers. + backend._estimate_compute_buffer_bytes = lambda **kwargs: 1 + backend._mtp_draft_kv_bytes = lambda *args, **kwargs: 0 + backend._estimate_mtp_overhead_bytes = lambda *args, **kwargs: int(drafter_gb * gb) + backend._fit_context_to_vram = lambda requested, *args, **kwargs: requested + backend._select_gpus = lambda *args, **kwargs: ([0], False) + backend._select_gpus_split_aware = lambda *args, **kwargs: ([0], False) + backend.probe_server_capabilities = lambda _binary = None: { + "supports_dspark": True, + "spec_draft_n_max_flag": "--spec-draft-n-max", + } + return backend, gguf, sidecar + + +def test_auto_drops_the_drafter_when_only_the_target_fits(tmp_path): + """Model fits, drafter does not: Auto keeps the context and runs without it. + + The alternative today is a silently smaller context (or --fit offload, where + decode collapses), paid for a speed option the user never asked for. + """ + backend, gguf, sidecar = _tight_vram_backend(tmp_path, drafter_gb = 12.0) + + result = _launch( + backend, + gguf, + dspark_draft_path = str(sidecar), + speculative_type = "auto", + n_ctx = 8192, + ) + + cmd = result["cmd"] + assert "--model-draft" not in cmd + assert "draft-dspark" not in cmd + assert cmd[cmd.index("-c") + 1] == "8192" + assert backend.spec_fallback_reason == "drafter_no_vram" + # Names the drafter Auto had resolved, so the notice does not read "MTP", and + # keeps the resolved path so a repeat Apply dedupes instead of relaunching. + assert backend.spec_drafter_kind == "dspark" + assert backend.mtp_draft_path == str(sidecar) + + +def test_auto_keeps_a_drafter_that_fits(tmp_path): + """The drop is scoped to the shortfall: with room for both, nothing changes.""" + backend, gguf, sidecar = _tight_vram_backend(tmp_path, drafter_gb = 1.5) + + result = _launch( + backend, + gguf, + dspark_draft_path = str(sidecar), + speculative_type = "auto", + n_ctx = 8192, + ) + + cmd = result["cmd"] + assert cmd[cmd.index("--model-draft") + 1] == str(sidecar) + assert cmd[cmd.index("--spec-type") + 1] == "draft-dspark" + assert backend.spec_fallback_reason is None + + +def test_forcing_the_drafter_overrides_the_vram_drop(tmp_path): + """Only Auto is second-guessed. An explicit choice launches the drafter and + lets the existing context reduction pay for it.""" + backend, gguf, sidecar = _tight_vram_backend(tmp_path, drafter_gb = 12.0) + + result = _launch( + backend, + gguf, + dspark_draft_path = str(sidecar), + speculative_type = "dspark", + n_ctx = 8192, + ) + + cmd = result["cmd"] + assert cmd[cmd.index("--model-draft") + 1] == str(sidecar) + assert cmd[cmd.index("--spec-type") + 1] == "draft-dspark" + assert backend.spec_fallback_reason is None + + +def test_an_embedded_mtp_head_is_dropped_too(tmp_path): + """No sidecar file to blank, so the drop has to reach the flags themselves. + + An embedded head still costs a draft KV and a verify graph, and the fit + reserved neither; emitting --spec-type draft-mtp anyway would OOM the load. + """ + backend, gguf, _sidecar = _tight_vram_backend(tmp_path, drafter_gb = 12.0) + backend._read_gguf_metadata = lambda _path: setattr(backend, "_nextn_predict_layers", 1) + backend.probe_server_capabilities = lambda _binary = None: { + "mtp_token": "draft-mtp", + "supports_ngram_mod": True, + "spec_draft_n_max_flag": "--spec-draft-n-max", + } + + result = _launch(backend, gguf, speculative_type = "auto", n_ctx = 8192) + + cmd = result["cmd"] + assert "draft-mtp" not in cmd + assert cmd[cmd.index("--spec-type") + 1] == "ngram-mod" + assert cmd[cmd.index("-c") + 1] == "8192" + assert backend.spec_fallback_reason == "drafter_no_vram" diff --git a/studio/frontend/src/features/chat/chat-settings-sheet.tsx b/studio/frontend/src/features/chat/chat-settings-sheet.tsx index 0b72ec3e2f..c56cffd784 100644 --- a/studio/frontend/src/features/chat/chat-settings-sheet.tsx +++ b/studio/frontend/src/features/chat/chat-settings-sheet.tsx @@ -425,6 +425,8 @@ function specFallbackMessage({ switch (reason) { case "mla_mtp_disabled": return "MTP is disabled by default for this model architecture because it currently runs slower than standard decoding. Choose MTP in the model picker to force it."; + case "drafter_no_vram": + return `This model fits in VRAM but its ${drafter} drafter does not, so Auto kept your context length and is running without speculative decoding. Choose ${drafter} in Settings to force it, at a smaller context.`; case "runtime_error": return `${drafter} could not start for this model on the installed llama.cpp build, so it is running without speculative decoding.`; case "drafter_not_found": diff --git a/studio/frontend/src/features/chat/types/api.ts b/studio/frontend/src/features/chat/types/api.ts index dbfeec3f9c..1dec2ff00b 100644 --- a/studio/frontend/src/features/chat/types/api.ts +++ b/studio/frontend/src/features/chat/types/api.ts @@ -352,6 +352,9 @@ export interface InferenceStatusResponse { * "binary_no_mtp" / "binary_outdated" -> updating llama.cpp would re-enable * it; "runtime_error" -> the current build could not run it; * "drafter_not_found" -> its MTP or DSpark sidecar was unavailable; + * "drafter_no_vram" -> an Auto-mode fit downgrade: the model pins on GPU but + * the drafter's reserve does not, and Auto keeps the context rather than + * shrink it (choose the drafter in Settings to force it); * "mla_mtp_disabled" -> an Auto-mode policy downgrade for MLA models * (GLM-5.2 et al.) whose llama.cpp MTP path is slower than no speculation * (updating won't help; choose MTP in Settings to force it). Null otherwise. From 21dee6573658555daaacfc4c1cd6d191e5e14c4e Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Tue, 11 Aug 2026 13:28:08 +0000 Subject: [PATCH 02/25] Do not charge a repository drafter the extras replace, nor a partial DFlash shard set --- studio/backend/routes/inference.py | 33 +++++++--- .../tests/test_chat_load_during_training.py | 62 +++++++++++++++++++ .../backend/utils/models/drafters/budget.py | 7 +++ 3 files changed, 92 insertions(+), 10 deletions(-) diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 055bf9b4d0..b1544c7ba3 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5863,12 +5863,24 @@ def _estimate_gguf_required_gb( ) _spec_mode = _canonicalize_spec_mode(speculative_type) or "auto" + _extra_args_own_spec = _extra_args_set_spec_type(llama_extra_args) + # Extras that own --spec-type AND name their own --model-draft end + # _build_speculative_flags before any mode branch, so Studio's sidecar is + # neither fetched nor launched: the drafter that becomes resident is theirs, + # already charged below as _extras_bytes, and billing both refuses a load that + # fits. Both halves matter. Owning the spec type alone keeps the conservative + # charge, since the guard protects a running training job and a drafter can + # still arrive by a route this cannot see. Applies to every kind, hence one flag. + _extras_own_drafter = bool( + _extra_args_own_spec and _extra_args_mtp_draft_path(llama_extra_args, env = {}) + ) _forced_dspark = bool( - _spec_mode == "dspark" or _extra_args_requests_dspark(llama_extra_args, env = {}) + (_spec_mode == "dspark" or _extra_args_requests_dspark(llama_extra_args, env = {})) + and not _extras_own_drafter ) # Auto loads the sidecar whenever the model has one, so size it there too # or the guard admits a load 11 GB larger than it estimated. - _auto_dspark = _spec_mode == "auto" + _auto_dspark = _spec_mode == "auto" and not _extras_own_drafter _dspark_capable = True if _forced_dspark or _auto_dspark: # Gate on the same answer the loader uses: _download_dspark skips the @@ -5892,13 +5904,12 @@ def _estimate_gguf_required_gb( # DFlash: same shape as DSpark above, and Auto sizes it for the same # reason. The sidecar is ~1.5 GiB rather than ~11 GB, but a guard that # protects a running training job still has to charge for it. - # Extra args owning --spec-type end _build_speculative_flags before any mode - # branch, so neither forced nor Auto reaches the sidecar and charging it refuses - # a load for nothing. Extras asking for draft-dflash themselves still pay. - _extra_args_own_spec = _extra_args_set_spec_type(llama_extra_args) _forced_dflash = bool( - _extra_args_requests_dflash(llama_extra_args, env = {}) - or (_spec_mode == "dflash" and not _extra_args_own_spec) + ( + _extra_args_requests_dflash(llama_extra_args, env = {}) + or (_spec_mode == "dflash" and not _extra_args_own_spec) + ) + and not _extras_own_drafter ) _auto_dflash = _spec_mode == "auto" and not _extra_args_own_spec _dflash_capable = True @@ -5919,8 +5930,10 @@ def _estimate_gguf_required_gb( # which loads no drafter at all, so charging the MTP one would refuse a load # that fits. Auto is different: it falls through to the MTP branch, and keeps # its charge. - _charge_no_drafter = (_forced_dspark and not _dspark_capable) or ( - _forced_dflash and not _dflash_capable + _charge_no_drafter = ( + _extras_own_drafter + or (_forced_dspark and not _dspark_capable) + or (_forced_dflash and not _dflash_capable) ) def _same_file_key(p: str) -> str: diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index 278ff6000b..9552053333 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -1571,6 +1571,68 @@ def _companion_bytes(siblings, **kwargs): # An explicit DFlash request is not the Auto race and still pays for it. self.assertEqual(_companion_bytes(both, include_dflash = True), 400) + def test_extras_owning_the_spec_type_are_not_charged_the_repo_sidecar(self): + """A caller who sets --spec-type ends _build_speculative_flags before any + mode branch, so no repository sidecar of any kind is fetched or launched. + Only the drafter their --model-draft names becomes resident, and that is + charged separately; billing the repo's on top is a 409 for a load that fits.""" + import utils.models.model_config as mc + + cfg = SimpleNamespace( + gguf_file = None, + gguf_mmproj_file = None, + gguf_mtp_file = None, + gguf_dspark_file = None, + gguf_dflash_file = None, + gguf_hf_repo = "org/repo", + gguf_variant = "Q4_K_M", + ) + variant = SimpleNamespace(quant = "Q4_K_M", size_bytes = 1024**3) + for extras in ( + ["--spec-type", "draft-dflash", "--model-draft", "/tmp/d.gguf"], + ["--spec-type", "draft-dspark", "--model-draft", "/tmp/d.gguf"], + ): + with ( + patch.object( + mc, "list_gguf_variants", lambda repo, hf_token = None: ([variant], False) + ), + patch.object(self.route, "_remote_gguf_companion_bytes", return_value = 0) as comp, + self._dflash_capable(), + ): + self.route._estimate_gguf_required_gb( + cfg, speculative_type = "auto", llama_extra_args = extras + ) + self.assertFalse(comp.call_args.kwargs["include_dflash"], extras) + self.assertFalse(comp.call_args.kwargs["include_dspark"], extras) + self.assertFalse(comp.call_args.kwargs["include_mtp"], extras) + + def test_a_partial_dflash_shard_set_is_not_charged(self): + """The fetch refuses a family whose encoded shard count is short, so a + listing caught mid-publication must not be billed for its listed half: + the DSpark path already filters on this, and the two have to agree.""" + partial = [ + SimpleNamespace(rfilename = "dflash-kquant-00001-of-00002.gguf", size = 400), + ] + whole = partial + [ + SimpleNamespace(rfilename = "dflash-kquant-00002-of-00002.gguf", size = 300), + ] + + def _companion_bytes(siblings): + with patch( + "huggingface_hub.model_info", + return_value = SimpleNamespace(siblings = siblings), + ): + return self.route._remote_gguf_companion_bytes( + "org/repo", + hf_token = None, + include_mmproj = False, + include_mtp = False, + include_dflash = True, + ) + + self.assertEqual(_companion_bytes(partial), 0) + self.assertEqual(_companion_bytes(whole), 700) + def test_auto_tells_the_companion_sizing_that_dspark_comes_first(self): """The remote branch is where both kinds can be asked for at once, so it is the caller that has to pass the loader's Auto rule down.""" diff --git a/studio/backend/utils/models/drafters/budget.py b/studio/backend/utils/models/drafters/budget.py index f89d4e4f53..cdc47b3672 100644 --- a/studio/backend/utils/models/drafters/budget.py +++ b/studio/backend/utils/models/drafters/budget.py @@ -12,6 +12,8 @@ from typing import Callable, Mapping +from utils.models.drafters.common import split_listing_is_complete + def dflash_budget_bytes( sizes: Mapping[str, int], @@ -38,10 +40,15 @@ def dflash_budget_bytes( ``target_bytes`` drops what the fetch itself refuses: a drafter is a few layers of its target, so a set at least that large is an ordinary weight wearing the prefix. Zero means unknown and keeps every candidate. + + An incomplete split set is refused for the same reason: the fetch turns those + families away on the shard count, so charging their listed part is a 409 for a + load that fits, which is what a mid-publication listing looks like. """ totals = ( size + sum(sizes.get(shard, 0) for shard in extra_shards(sizes, name)) for name, size in sizes.items() + if split_listing_is_complete(sizes, name) ) return max( (total for total in totals if not target_bytes or total < target_bytes), From b6fb82330ac8a756cd472d46e0943fe4dad9c40a Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Tue, 11 Aug 2026 13:33:51 +0000 Subject: [PATCH 03/25] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/core/inference/llama_cpp.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 5434e578e5..5d99c9e462 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -11237,9 +11237,9 @@ def _probe_cc(ctx: int) -> int: if _ctx_wo > 0: _shared = _kv_bytes(_ctx_wo) + _probe_cc(_ctx_wo) _foot_wo = (_base_wo + _shared) / (1024 * 1024) - _foot_w = ( - _probe_base(True) + _shared + _mtp_bytes(_ctx_wo) - ) / (1024 * 1024) + _foot_w = (_probe_base(True) + _shared + _mtp_bytes(_ctx_wo)) / ( + 1024 * 1024 + ) if _foot_wo <= _budget_wo and _foot_w > _probe_budget(True): _spec_dropped_no_vram = True _mtp_will_engage = False From 4176b6059b20ecd9a4e7d5acbb0830bc53ba881d Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Tue, 11 Aug 2026 14:11:05 +0000 Subject: [PATCH 04/25] Fix the drafter drop: ngram capability, extras drafters, GPU subsets and tensor mode --- studio/backend/core/inference/llama_cpp.py | 185 +++++++++++------- studio/backend/routes/inference.py | 12 +- .../tests/test_chat_load_during_training.py | 56 +++++- .../backend/tests/test_llama_cpp_placement.py | 106 ++++++++++ 4 files changed, 290 insertions(+), 69 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 5d99c9e462..8a1c6c33d6 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -11163,10 +11163,18 @@ def _cc_split_extra(ctx: int) -> int: # for it with a smaller context (or a --fit offload, where decode # collapses ~3x) is the worse trade. An explicit dropdown / CLI / # extras choice is honored and keeps the shrink behaviour below. - # Priced over the WHOLE GPU set, the ceiling for any subset the fit - # then picks, at the context the target alone would get: so it drops - # only when no placement could have held both. Carried to the launch - # as drafter_no_vram, which downgrades the Auto emit to ngram-mod. + # Priced over the SAME ranked subsets the placement loop below + # walks, at the context the target alone would get on each: so it + # drops only when no placement could have held both. A whole-pool + # figure is not the ceiling it looks like, because a busy card adds + # ~nothing to _pool_budget_mib while still charging its pipeline + # overhead and its replicated compute buffer, which can condemn a + # drafter the single healthy GPU would have held. Carried to the + # launch as drafter_no_vram, which downgrades the Auto emit. + # Skipped under tensor parallelism: _plan_tensor_parallel reserves a + # per-device tensor buffer with its own context geometry, so these + # layer-split numbers are not that load's numbers, and a wrong answer + # here costs the user the drafter. TP keeps the behaviour it has. # The gate reads the user's choice, not _mtp_effective, which by # here is the kind Auto resolved and can no longer tell the two apart. if ( @@ -11175,14 +11183,30 @@ def _cc_split_extra(ctx: int) -> int: and not _user_mtp_via_extras and not _user_draft_via_extras and not _extra_args_set_spec_type(extra_args) + # A bare --model-draft / --spec-draft-hf sets neither of the + # two flags above (both key off an accumulated --spec-type), + # yet llama.cpp loads whatever that names regardless of the + # spec type: server load_model gates the draft model on + # has_dft(), i.e. "a draft path was given". Dropping here + # would release the reserve for a drafter the child still + # loads, and the load OOMs. It is also an explicit choice. + and not _extra_args_mtp_draft_path(extra_args, env = _spec_env) and not _draft_cpu_no_embedded + and not tensor_parallel and gpus and effective_ctx > 0 and self._can_estimate_kv() ): - _probe_gpus = len(gpus) - def _probe_base(drafter: bool) -> int: + def _probe_frac(drafter: bool) -> float: + # The flat fraction is the reserve whenever _mtp_bytes is 0. + return self._GPU_PIN_VRAM_FRACTION - ( + _MTP_VRAM_RESERVE_FRAC + if (drafter and (mtp_overhead_fn is None or _mtp_kv_unsized)) + else 0.0 + ) + + def _probe_base(drafter: bool, n: int) -> int: # model_size_fit's terms, before it is built below. _soft = self._CUDA_CONTEXT_RESERVE_BYTES if effective_is_vision and mmproj_size > 0: @@ -11193,66 +11217,83 @@ def _probe_base(drafter: bool) -> int: model_size + _compute_buffer_pipeline + _soft - + max(0, _probe_gpus - 1) * _pipeline_overhead_bytes + + max(0, n - 1) * _pipeline_overhead_bytes ) - def _probe_budget(drafter: bool) -> float: - # The flat fraction is the reserve whenever _mtp_bytes is 0. - return _pool_budget_mib( - gpus, - self._GPU_PIN_VRAM_FRACTION - - ( - _MTP_VRAM_RESERVE_FRAC - if (drafter and (mtp_overhead_fn is None or _mtp_kv_unsized)) - else 0.0 - ), - ) - - def _probe_cc(ctx: int) -> int: - return _cc_bytes(ctx, _probe_gpus) - - _base_wo, _budget_wo = _probe_base(False), _probe_budget(False) - # Explicit context is honored verbatim, so that is what the - # drafter has to fit alongside; Auto gets its own best cap. - _ctx_wo = ( - effective_ctx - if explicit_ctx - else self._fit_context_to_vram( - effective_ctx, - _budget_wo, - _base_wo, - cache_type_kv, - swa_full = swa_full, - n_parallel = n_parallel, - kv_unified = planned_kv_unified, - n_ubatch = _effective_ubatch, - flash_attn = planned_flash_attn, - mtp_engaged = False, - mtp_overhead_fn = None, - compute_ctx_bytes_fn = _probe_cc, - budget_frac = 1.0, - total_mib = None, - ) + # Fewest GPUs first, ranked by usable VRAM: the same order and + # the same budget the auto placement loop uses, so the answer is + # about placements that loop could actually choose. + _probe_ranked = sorted( + gpus, + key = lambda g: _gpu_usable(g, _probe_frac(False)), + reverse = True, ) - if _ctx_wo > 0: - _shared = _kv_bytes(_ctx_wo) + _probe_cc(_ctx_wo) - _foot_wo = (_base_wo + _shared) / (1024 * 1024) - _foot_w = (_probe_base(True) + _shared + _mtp_bytes(_ctx_wo)) / ( - 1024 * 1024 + _target_fits_somewhere = False + _both_fit_somewhere = False + _probe_ctx = 0 + _probe_need = _probe_have = 0.0 + for _n in range(1, len(_probe_ranked) + 1): + _subset = _probe_ranked[:_n] + _cc_n = lambda c, _k = _n: _cc_bytes(c, _k) + _base_wo = _probe_base(False, _n) + _budget_wo = _pool_budget_mib(_subset, _probe_frac(False)) + # Explicit context is honored verbatim, so that is what the + # drafter has to fit alongside; Auto gets its own best cap. + _ctx_wo = ( + effective_ctx + if explicit_ctx + else self._fit_context_to_vram( + effective_ctx, + _budget_wo, + _base_wo, + cache_type_kv, + swa_full = swa_full, + n_parallel = n_parallel, + kv_unified = planned_kv_unified, + n_ubatch = _effective_ubatch, + flash_attn = planned_flash_attn, + mtp_engaged = False, + mtp_overhead_fn = None, + compute_ctx_bytes_fn = _cc_n, + budget_frac = 1.0, + total_mib = None, + ) ) - if _foot_wo <= _budget_wo and _foot_w > _probe_budget(True): - _spec_dropped_no_vram = True - _mtp_will_engage = False - logger.warning( - "Speculative decoding disabled for this load: the model " - "fits in VRAM at context %d but its drafter does not " - "(needs %.1f GB of a %.1f GB budget). Auto keeps the " - "context rather than shrink it for a speed option. " - "Select the drafter in Settings to force it.", + if _ctx_wo <= 0: + continue + _shared = _kv_bytes(_ctx_wo) + _cc_n(_ctx_wo) + _foot_wo = (_base_wo + _shared) / (1024 * 1024) + if _foot_wo > _budget_wo: + continue + _foot_w = ( + _probe_base(True, _n) + _shared + _mtp_bytes(_ctx_wo) + ) / (1024 * 1024) + _budget_w = _pool_budget_mib(_subset, _probe_frac(True)) + if not _target_fits_somewhere: + # The placement this reports on: the first subset that + # holds the target, which is the one the loop would pick. + _target_fits_somewhere = True + _probe_ctx, _probe_need, _probe_have = ( _ctx_wo, - _foot_w / 1024, - _probe_budget(True) / 1024, + _foot_w, + _budget_w, ) + if _foot_w <= _budget_w: + _both_fit_somewhere = True + break + if _target_fits_somewhere and not _both_fit_somewhere: + _spec_dropped_no_vram = True + _mtp_will_engage = False + logger.warning( + "Speculative decoding disabled for this load: the model " + "fits in VRAM at context %d but its drafter does not " + "(needs %.1f GB of a %.1f GB budget, on any GPU subset). " + "Auto keeps the context rather than shrink it for a speed " + "option. Select the drafter in Settings to force it.", + _probe_ctx, + _probe_need / 1024, + _probe_have / 1024, + ) # Flat MTP reserve fraction: used only as the fallback when the # byte-accurate mtp_overhead_fn can't size the draft KV (dims @@ -13983,12 +14024,24 @@ def _fallback_drafter_not_found() -> None: # spec-off when the build lacks it. The drafter paths stay recorded, so # the UI still names the kind and a repeat Apply still dedupes. self._spec_fallback_reason = "drafter_no_vram" - logger.info( - "Auto: the drafter does not fit in VRAM alongside the model at this " - "context, so it is dropped. Choose it in the Speculative Decoding " - "dropdown to force it at a smaller context." - ) - _emit_ngram_mod() + if caps.get("supports_ngram_mod"): + logger.info( + "Auto: the drafter does not fit in VRAM alongside the model at this " + "context, so it is dropped and ngram-mod (zero-VRAM) takes its " + "place. Choose the drafter in the Speculative Decoding dropdown to " + "force it at a smaller context." + ) + _emit_ngram_mod() + else: + # spec-off: --spec-type ngram-mod is not a value this build's enum + # carries, and llama-server aborts on one it cannot parse rather than + # ignoring it. Mirrors the MLA and sub-3B branches below. + logger.info( + "Auto: the drafter does not fit in VRAM alongside the model at this " + "context, so speculative decoding is disabled (this llama-server " + "does not advertise ngram-mod). Choose the drafter in the " + "Speculative Decoding dropdown to force it at a smaller context." + ) elif dspark_draft_path and caps.get("supports_dspark"): # DSpark first: load_model only hands a sidecar down once it has one # this binary can launch, and it beats every other Auto outcome for diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index b1544c7ba3..0faca419ad 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5871,8 +5871,18 @@ def _estimate_gguf_required_gb( # fits. Both halves matter. Owning the spec type alone keeps the conservative # charge, since the guard protects a running training job and a drafter can # still arrive by a route this cannot see. Applies to every kind, hence one flag. + # Only a LOCAL file, because the suppression's whole premise is that the + # drafter is "already charged below as _extras_bytes" and that charge is + # itself gated on Path(...).is_file(). --spec-draft-hf / -hfd names an HF + # repo id, which llama-server downloads and loads all the same (its + # has_dft() only asks whether a draft path was given), so suppressing on + # one would leave a multi-GB resident drafter charged nowhere and let the + # guard admit a load that evicts the training job it protects. + _extras_own_draft_path = _extra_args_mtp_draft_path(llama_extra_args, env = {}) _extras_own_drafter = bool( - _extra_args_own_spec and _extra_args_mtp_draft_path(llama_extra_args, env = {}) + _extra_args_own_spec + and _extras_own_draft_path + and Path(_extras_own_draft_path).is_file() ) _forced_dspark = bool( (_spec_mode == "dspark" or _extra_args_requests_dspark(llama_extra_args, env = {})) diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index 9552053333..de019505c5 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -1588,9 +1588,17 @@ def test_extras_owning_the_spec_type_are_not_charged_the_repo_sidecar(self): gguf_variant = "Q4_K_M", ) variant = SimpleNamespace(quant = "Q4_K_M", size_bytes = 1024**3) + import tempfile + + _tmp = tempfile.TemporaryDirectory() + self.addCleanup(_tmp.cleanup) + # A real file: the suppression is only sound for a drafter _extras_bytes + # can actually charge, and that charge is gated on Path(...).is_file(). + _draft = Path(_tmp.name) / "d.gguf" + _draft.write_bytes(b"x" * 512) for extras in ( - ["--spec-type", "draft-dflash", "--model-draft", "/tmp/d.gguf"], - ["--spec-type", "draft-dspark", "--model-draft", "/tmp/d.gguf"], + ["--spec-type", "draft-dflash", "--model-draft", str(_draft)], + ["--spec-type", "draft-dspark", "--model-draft", str(_draft)], ): with ( patch.object( @@ -1606,6 +1614,50 @@ def test_extras_owning_the_spec_type_are_not_charged_the_repo_sidecar(self): self.assertFalse(comp.call_args.kwargs["include_dspark"], extras) self.assertFalse(comp.call_args.kwargs["include_mtp"], extras) + def test_a_remote_extras_drafter_still_pays_the_conservative_charge(self): + """--spec-draft-hf names an HF repo, not a file, so _extras_bytes charges + nothing for it while llama-server still downloads and loads it (its + has_dft() only asks whether a draft path was given). Suppressing the + repository sidecar on top would leave the resident drafter charged + nowhere and admit a load that evicts the training job this guard protects.""" + import utils.models.model_config as mc + from core.inference.llama_cpp import LlamaCppBackend + + cfg = SimpleNamespace( + gguf_file = None, + gguf_mmproj_file = None, + gguf_mtp_file = None, + gguf_dspark_file = None, + gguf_dflash_file = None, + gguf_hf_repo = "org/repo", + gguf_variant = "Q4_K_M", + ) + variant = SimpleNamespace(quant = "Q4_K_M", size_bytes = 1024**3) + for extras, kind in ( + (["--spec-type", "draft-dspark", "--spec-draft-hf", "org/drafter"], "include_dspark"), + (["--spec-type", "draft-dflash", "-hfd", "org/drafter"], "include_dflash"), + ): + with ( + patch.object( + mc, "list_gguf_variants", lambda repo, hf_token = None: ([variant], False) + ), + patch.object(self.route, "_remote_gguf_companion_bytes", return_value = 0) as comp, + patch.object( + LlamaCppBackend, + "probe_server_capabilities", + classmethod( + lambda cls, binary = None: { + "supports_dspark": True, + "supports_dflash": True, + } + ), + ), + ): + self.route._estimate_gguf_required_gb( + cfg, speculative_type = "auto", llama_extra_args = extras + ) + self.assertTrue(comp.call_args.kwargs[kind], extras) + def test_a_partial_dflash_shard_set_is_not_charged(self): """The fetch refuses a family whose encoded shard count is short, so a listing caught mid-publication must not be billed for its listed half: diff --git a/studio/backend/tests/test_llama_cpp_placement.py b/studio/backend/tests/test_llama_cpp_placement.py index d4b74f7c9c..afcf3f46d0 100644 --- a/studio/backend/tests/test_llama_cpp_placement.py +++ b/studio/backend/tests/test_llama_cpp_placement.py @@ -476,3 +476,109 @@ def test_an_embedded_mtp_head_is_dropped_too(tmp_path): assert cmd[cmd.index("--spec-type") + 1] == "ngram-mod" assert cmd[cmd.index("-c") + 1] == "8192" assert backend.spec_fallback_reason == "drafter_no_vram" + + +def test_the_vram_drop_does_not_emit_ngram_mod_on_a_build_without_it(tmp_path): + """`ngram-mod` is a value in llama.cpp's --spec-type enum, so a build that + predates it aborts on the flag instead of ignoring it. The MLA and sub-3B + fallbacks gate on the capability for exactly that reason; this one has to too, + or the drop turns a slower load into a load that never starts.""" + backend, gguf, _sidecar = _tight_vram_backend(tmp_path, drafter_gb = 12.0) + backend._read_gguf_metadata = lambda _path: setattr(backend, "_nextn_predict_layers", 1) + backend.probe_server_capabilities = lambda _binary = None: { + "mtp_token": "draft-mtp", + "supports_ngram_mod": False, + "spec_draft_n_max_flag": "--spec-draft-n-max", + } + + result = _launch(backend, gguf, speculative_type = "auto", n_ctx = 8192) + + cmd = result["cmd"] + assert "ngram-mod" not in cmd + assert "--spec-type" not in cmd + assert "draft-mtp" not in cmd + assert cmd[cmd.index("-c") + 1] == "8192" + assert backend.spec_fallback_reason == "drafter_no_vram" + + +def test_a_standalone_model_draft_in_extras_is_not_auto_dropped(tmp_path): + """--model-draft alone sets no --spec-type, so neither extras probe fires, but + llama-server loads whatever it names regardless of the spec type (load_model + gates the draft model on has_dft(), i.e. "a draft path was given"). Dropping it + releases the reserve for a drafter the child still loads, and it is an explicit + user choice besides.""" + backend, gguf, sidecar = _tight_vram_backend(tmp_path, drafter_gb = 12.0) + user_draft = tmp_path / "my-drafter.gguf" + user_draft.write_bytes(b"draft") + + result = _launch( + backend, + gguf, + dspark_draft_path = str(sidecar), + speculative_type = "auto", + n_ctx = 8192, + extra_args = ["--model-draft", str(user_draft)], + ) + + cmd = result["cmd"] + assert "ngram-mod" not in cmd + assert cmd[cmd.index("--spec-type") + 1] == "draft-dspark" + assert backend.spec_fallback_reason is None + + +def test_a_busy_second_gpu_does_not_condemn_a_drafter_the_first_one_holds(tmp_path): + """A whole-pool figure is not the ceiling it looks like. + + A card with almost nothing free adds ~0 to the pooled budget, but a two-GPU + layer split still charges its 1 GiB pipeline overhead, so pricing the drafter + over the whole pool can reject one the single healthy GPU holds comfortably. + The probe walks the same ranked subsets the placement loop does, so the 1-GPU + placement it would actually pick is the one that decides. + """ + backend, gguf, sidecar = _tight_vram_backend(tmp_path, drafter_gb = 5.0) + # GPU1 is in use by something else: 800 MiB free of 24 GiB. + backend._get_gpu_memory = lambda _binary = None: [ + (0, 24_576, 24_576), + (1, 800, 24_576), + ] + backend._get_gpu_free_memory = lambda _binary = None: [(0, 24_576), (1, 800)] + + result = _launch( + backend, + gguf, + dspark_draft_path = str(sidecar), + speculative_type = "auto", + n_ctx = 8192, + ) + + cmd = result["cmd"] + assert cmd[cmd.index("--model-draft") + 1] == str(sidecar) + assert cmd[cmd.index("--spec-type") + 1] == "draft-dspark" + assert backend.spec_fallback_reason is None + + +def test_tensor_parallel_keeps_its_own_sizing(tmp_path): + """_plan_tensor_parallel reserves a per-device tensor buffer on geometry this + layer-split probe does not model, so under tensor mode the probe stands down + rather than decide the drafter's fate on numbers that are not that load's.""" + # Two cards that only hold the 16 GB target together, so the layer-split + # probe would condemn the drafter if it were allowed to answer here. + backend, gguf, sidecar = _tight_vram_backend(tmp_path, drafter_gb = 12.0) + backend._get_gpu_memory = lambda _binary = None: [ + (0, 12_288, 12_288), + (1, 12_288, 12_288), + ] + backend._get_gpu_free_memory = lambda _binary = None: [(0, 12_288), (1, 12_288)] + backend._tensor_split_aborts = lambda *args, **kwargs: False + + result = _launch( + backend, + gguf, + dspark_draft_path = str(sidecar), + speculative_type = "auto", + tensor_parallel = True, + n_ctx = 8192, + ) + + assert backend.spec_fallback_reason != "drafter_no_vram" + assert "--model-draft" in result["cmd"] From d9b7e7c7e820348b6382b7e4814771a3f249b8e1 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Tue, 11 Aug 2026 14:12:18 +0000 Subject: [PATCH 05/25] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/core/inference/llama_cpp.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 8a1c6c33d6..3cc391fa60 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -11265,9 +11265,9 @@ def _probe_base(drafter: bool, n: int) -> int: _foot_wo = (_base_wo + _shared) / (1024 * 1024) if _foot_wo > _budget_wo: continue - _foot_w = ( - _probe_base(True, _n) + _shared + _mtp_bytes(_ctx_wo) - ) / (1024 * 1024) + _foot_w = (_probe_base(True, _n) + _shared + _mtp_bytes(_ctx_wo)) / ( + 1024 * 1024 + ) _budget_w = _pool_budget_mib(_subset, _probe_frac(True)) if not _target_fits_somewhere: # The placement this reports on: the first subset that From 9daf6ba6b4a806ab6eeb8fef30dab7abf1979f23 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Tue, 11 Aug 2026 14:45:36 +0000 Subject: [PATCH 06/25] Probe the layer load a tensor request falls back to, and cap it per device --- studio/backend/core/inference/llama_cpp.py | 225 +++++++++++------- studio/backend/tests/test_compute_buffer.py | 6 +- .../backend/tests/test_llama_cpp_placement.py | 88 +++++++ .../src/features/chat/chat-settings-sheet.tsx | 4 +- 4 files changed, 234 insertions(+), 89 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 3cc391fa60..212cf9f46a 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -11158,6 +11158,98 @@ def _cc_split_extra(ctx: int) -> int: explicit_ctx = requested_ctx > 0 _draft_cpu_no_embedded = _draft_on_cpu and not self._nextn_predict_layers + # The two tensor -> layer downgrades that depend on nothing the + # drafter probe below decides run here, BEFORE it: the probe is + # gated on `not tensor_parallel`, and a load that ends up layer-split + # is a load the probe has to answer for (the layer planner reserves + # the Auto drafter and pays for it in context). Both are knowable + # this early -- a recorded abort is a session lookup, and the usable + # GPU count needs only the tensor compute-buffer reserve -- and + # running them first also hands the probe the restored (possibly + # quantized) KV type the layer load will actually use. + def _restore_after_tensor_downgrade(): + # Restore the quantized KV + extras tensor dropped (layer + # split supports them), minus --split-mode. + nonlocal cache_type_kv, _cache_type_from_env, extra_args + if _tensor_dropped_cache_type_kv is not None: + cache_type_kv = _tensor_dropped_cache_type_kv + _cache_type_from_env = False + extra_args = strip_split_mode_only( + _tensor_dropped_extra_args + if _tensor_dropped_extra_args is not None + else extra_args + ) + + # The route fallback retry is tensor-off; keep it multi-GPU. + if preserve_multi_gpu_on_layer: + _layer_min_gpus = max(_layer_min_gpus, len(gpus)) + + if tensor_parallel and self._tensor_split_aborts(binary, model_identifier): + # Aborted on tensor for this model this session (#6415); skip + # tensor upfront, layer split serves it. + logger.info( + "Tensor parallelism skipped: this llama.cpp build aborted " + "on --split-mode tensor for this model earlier this " + "session; using layer split across %d GPU(s).", + len(gpus), + ) + tensor_parallel = False + # Keep the multi-GPU request (gated on it, not the cache). + _layer_min_gpus = max(_layer_min_gpus, len(gpus)) + _restore_after_tensor_downgrade() + + # Tensor mode replicates a compute buffer on every GPU, so drop + # GPUs below that reserve from the set up front (gpu_indices + # becomes the CUDA_VISIBLE_DEVICES mask, fully excluding them). + tp_gpus = gpus + # Manual mode owns the layer count and context, so it skips + # the memory-based planner; its toggle still emits + # --split-mode tensor below (split by free VRAM, or by the + # Split ratio if set). auto plans here. + plan_tp = tensor_parallel and gpu_memory_mode != "manual" + if plan_tp: + # Deterministic per-device compute buffer (replicated on + # every device in tensor mode); flat fallback when dims + # are unavailable. _plan_tensor_parallel uses the same. + _tp_reserve_bytes = self._estimate_compute_buffer_bytes( + n_ubatch = _effective_ubatch, + n_parallel = n_parallel, + per_device_tensor = True, + ) + reserve_mib = ( + _tp_reserve_bytes // (1024 * 1024) + if _tp_reserve_bytes > 0 + else self._TENSOR_PARALLEL_BUFFER_RESERVE_MIB + ) + # Admit by usable budget (free - (1-frac)*total), not raw + # free: a partly-used big card can clear the reserve on raw + # free yet have no budget left. + tp_gpus = [g for g in gpus if _gpu_usable(g) >= reserve_mib] + + if plan_tp and len(tp_gpus) < 2: + # Tensor parallelism needs >= 2 usable GPUs. On a single + # GPU --split-mode tensor is a no-op; with 0 GPUs (CPU-only + # or probe failed) it must not reach llama-server; and a + # GPU below the buffer reserve can't participate. Drop the + # flag and fall through to normal layer/CPU allocation. + logger.info( + "Tensor parallelism requested but only %d of %d GPU(s) " + "have enough free VRAM for the compute buffer; " + "ignoring (needs >= 2).", + len(tp_gpus), + len(gpus), + ) + tensor_parallel = False + # GPUs below tensor's compute-buffer reserve can still do layer + # split, so keep multi-GPU (mirrors the budget/geometry drops); + # _select_gpus caps unusable cards. + if len(gpus) >= 2: + _layer_min_gpus = max(_layer_min_gpus, len(gpus)) + # Layer split supports a quantized KV the tensor attempt + # dropped; restore the original cache type + extras (minus + # --split-mode) so the layer launch re-emits them. + _restore_after_tensor_downgrade() + # When the target pins on GPU but the drafter's reserve is what tips # it over, Auto drops the drafter: it only buys speed, and paying # for it with a smaller context (or a --fit offload, where decode @@ -11175,6 +11267,18 @@ def _cc_split_extra(ctx: int) -> int: # per-device tensor buffer with its own context geometry, so these # layer-split numbers are not that load's numbers, and a wrong answer # here costs the user the drafter. TP keeps the behaviour it has. + # `tensor_parallel` here already reflects the two downgrades hoisted + # above (recorded abort, fewer than two usable GPUs), so a request + # that ends up layer-split for either reason IS probed. One + # downgrade still escapes: the pooled tensor weight-budget check + # below, which prices _soft_overhead and therefore _mtp_reserves_gpu + # -- both derived from the _mtp_will_engage this probe may clear. + # Running it first would be circular (dropping the drafter shrinks + # the requirement, so the load could stay on tensor with the drafter + # already gone), so a tensor request that falls back to layer split + # purely because the pooled budget cannot hold weights + MTP reserve + # + per-device buffers keeps today's behaviour: no probe, and the + # layer planner pays for the drafter in context as before. # The gate reads the user's choice, not _mtp_effective, which by # here is the kind Auto resolved and can no longer tell the two apart. if ( @@ -11265,9 +11369,41 @@ def _probe_base(drafter: bool, n: int) -> int: _foot_wo = (_base_wo + _shared) / (1024 * 1024) if _foot_wo > _budget_wo: continue - _foot_w = (_probe_base(True, _n) + _shared + _mtp_bytes(_ctx_wo)) / ( - 1024 * 1024 - ) + # The pooled figure hides the compute buffer every device + # replicates: on a heterogeneous split the weakest card can + # be unable to hold the context the pool priced, and the + # placement loop caps to what it does hold. Charging the + # drafter at the uncapped context condemns it at a context + # this load can never reach. Same reserve expression and + # same cap the auto-context loop below applies (_reserve_at + # / _every_gpu_holds_reserve / _cap_ctx_to_per_device_reserve), + # so the two cannot disagree. Auto only: an explicit context + # is honored verbatim, never capped, and overflows to --fit. + if not explicit_ctx: + _usable_wo = [ + _gpu_usable(g, _probe_frac(False)) for g in _subset + ] + _probe_reserve_at = lambda c, _k = _n: ( + (_pipeline_overhead_bytes if _k > 1 else 0) + + _cc_bytes(c, _k) // _k + ) + if not self._every_gpu_holds_reserve( + _usable_wo, _probe_reserve_at(_ctx_wo) + ): + _ctx_wo = self._cap_ctx_to_per_device_reserve( + _ctx_wo, _usable_wo, _probe_reserve_at + ) + if _ctx_wo <= 0: + continue + # Every pooled term shrinks with the context, so this + # cannot newly fail; re-price rather than lean on it. + _shared = _kv_bytes(_ctx_wo) + _cc_n(_ctx_wo) + _foot_wo = (_base_wo + _shared) / (1024 * 1024) + if _foot_wo > _budget_wo: + continue + _foot_w = ( + _probe_base(True, _n) + _shared + _mtp_bytes(_ctx_wo) + ) / (1024 * 1024) _budget_w = _pool_budget_mib(_subset, _probe_frac(True)) if not _target_fits_somewhere: # The placement this reports on: the first subset that @@ -11331,89 +11467,6 @@ def _subset_model_size(n_gpus: int) -> int: # Unified-memory budget (0 off Apple Silicon) for the no-GPU Metal cap below. _apple_budget_mib = self._apple_metal_memory_budget_bytes() // (1024 * 1024) - def _restore_after_tensor_downgrade(): - # Restore the quantized KV + extras tensor dropped (layer - # split supports them), minus --split-mode. - nonlocal cache_type_kv, _cache_type_from_env, extra_args - if _tensor_dropped_cache_type_kv is not None: - cache_type_kv = _tensor_dropped_cache_type_kv - _cache_type_from_env = False - extra_args = strip_split_mode_only( - _tensor_dropped_extra_args - if _tensor_dropped_extra_args is not None - else extra_args - ) - - # The route fallback retry is tensor-off; keep it multi-GPU. - if preserve_multi_gpu_on_layer: - _layer_min_gpus = max(_layer_min_gpus, len(gpus)) - - if tensor_parallel and self._tensor_split_aborts(binary, model_identifier): - # Aborted on tensor for this model this session (#6415); skip - # tensor upfront, layer split serves it. - logger.info( - "Tensor parallelism skipped: this llama.cpp build aborted " - "on --split-mode tensor for this model earlier this " - "session; using layer split across %d GPU(s).", - len(gpus), - ) - tensor_parallel = False - # Keep the multi-GPU request (gated on it, not the cache). - _layer_min_gpus = max(_layer_min_gpus, len(gpus)) - _restore_after_tensor_downgrade() - - # Tensor mode replicates a compute buffer on every GPU, so drop - # GPUs below that reserve from the set up front (gpu_indices - # becomes the CUDA_VISIBLE_DEVICES mask, fully excluding them). - tp_gpus = gpus - # Manual mode owns the layer count and context, so it skips - # the memory-based planner; its toggle still emits - # --split-mode tensor below (split by free VRAM, or by the - # Split ratio if set). auto plans here. - plan_tp = tensor_parallel and gpu_memory_mode != "manual" - if plan_tp: - # Deterministic per-device compute buffer (replicated on - # every device in tensor mode); flat fallback when dims - # are unavailable. _plan_tensor_parallel uses the same. - _tp_reserve_bytes = self._estimate_compute_buffer_bytes( - n_ubatch = _effective_ubatch, - n_parallel = n_parallel, - per_device_tensor = True, - ) - reserve_mib = ( - _tp_reserve_bytes // (1024 * 1024) - if _tp_reserve_bytes > 0 - else self._TENSOR_PARALLEL_BUFFER_RESERVE_MIB - ) - # Admit by usable budget (free - (1-frac)*total), not raw - # free: a partly-used big card can clear the reserve on raw - # free yet have no budget left. - tp_gpus = [g for g in gpus if _gpu_usable(g) >= reserve_mib] - - if plan_tp and len(tp_gpus) < 2: - # Tensor parallelism needs >= 2 usable GPUs. On a single - # GPU --split-mode tensor is a no-op; with 0 GPUs (CPU-only - # or probe failed) it must not reach llama-server; and a - # GPU below the buffer reserve can't participate. Drop the - # flag and fall through to normal layer/CPU allocation. - logger.info( - "Tensor parallelism requested but only %d of %d GPU(s) " - "have enough free VRAM for the compute buffer; " - "ignoring (needs >= 2).", - len(tp_gpus), - len(gpus), - ) - tensor_parallel = False - # GPUs below tensor's compute-buffer reserve can still do layer - # split, so keep multi-GPU (mirrors the budget/geometry drops); - # _select_gpus caps unusable cards. - if len(gpus) >= 2: - _layer_min_gpus = max(_layer_min_gpus, len(gpus)) - # Layer split supports a quantized KV the tensor attempt - # dropped; restore the original cache type + extras (minus - # --split-mode) so the layer launch re-emits them. - _restore_after_tensor_downgrade() - if tensor_parallel and tp_gpus: # Pooled usable budget (after each device's compute buffer) # must hold the non-shrinkable footprint: weights + the MTP diff --git a/studio/backend/tests/test_compute_buffer.py b/studio/backend/tests/test_compute_buffer.py index a695013047..3a28061671 100644 --- a/studio/backend/tests/test_compute_buffer.py +++ b/studio/backend/tests/test_compute_buffer.py @@ -933,8 +933,10 @@ def test_wired_into_the_auto_context_loop(self): import inspect compact = "".join(inspect.getsource(LlamaCppBackend.load_model).split()) - # Native-context loop and the reduced-to-4096 fallback below it. - assert compact.count("ifnotself._every_gpu_holds_reserve(") == 2 + # Native-context loop, the reduced-to-4096 fallback below it, and the + # Auto drafter-drop probe above them, which caps to the same reserve so + # it cannot price a drafter at a context the weakest card never holds. + assert compact.count("ifnotself._every_gpu_holds_reserve(") == 3 # Gated on the chosen context, and only reachable after the pooled test. assert "_usable_mib=[_gpu_usable(g,pin_fraction)forginsubset]" in compact assert "(_gpu_usable(g,pin_fraction)forginsubset)," in compact diff --git a/studio/backend/tests/test_llama_cpp_placement.py b/studio/backend/tests/test_llama_cpp_placement.py index afcf3f46d0..04cad582c4 100644 --- a/studio/backend/tests/test_llama_cpp_placement.py +++ b/studio/backend/tests/test_llama_cpp_placement.py @@ -582,3 +582,91 @@ def test_tensor_parallel_keeps_its_own_sizing(tmp_path): assert backend.spec_fallback_reason != "drafter_no_vram" assert "--model-draft" in result["cmd"] + + +def test_a_tensor_request_that_aborted_before_is_probed_as_the_layer_load_it_is(tmp_path): + """A recorded --split-mode tensor abort downgrades the load to a layer split + before anything is planned, and the layer planner does reserve the Auto drafter + (paying for it in context). Gating the probe on the REQUESTED tensor flag would + hand that load the silent context cut the probe exists to prevent.""" + backend, gguf, sidecar = _tight_vram_backend(tmp_path, drafter_gb = 12.0) + backend._tensor_split_aborts = lambda *args, **kwargs: True + + result = _launch( + backend, + gguf, + dspark_draft_path = str(sidecar), + speculative_type = "auto", + tensor_parallel = True, + n_ctx = 8192, + ) + + cmd = result["cmd"] + assert "--split-mode" not in cmd + assert "--model-draft" not in cmd + assert cmd[cmd.index("-c") + 1] == "8192" + assert backend.spec_fallback_reason == "drafter_no_vram" + + +def test_a_single_gpu_tensor_request_is_probed_as_the_layer_load_it_is(tmp_path): + """Same shape, the commonest cause: tensor parallelism needs >= 2 usable GPUs, + so a one-card request is downgraded to a layer split and must be probed.""" + backend, gguf, sidecar = _tight_vram_backend(tmp_path, drafter_gb = 12.0) + backend._tensor_split_aborts = lambda *args, **kwargs: False + + result = _launch( + backend, + gguf, + dspark_draft_path = str(sidecar), + speculative_type = "auto", + tensor_parallel = True, + n_ctx = 8192, + ) + + cmd = result["cmd"] + assert "--split-mode" not in cmd + assert "--model-draft" not in cmd + assert cmd[cmd.index("-c") + 1] == "8192" + assert backend.spec_fallback_reason == "drafter_no_vram" + + +def test_the_probe_prices_the_drafter_at_a_context_the_weakest_card_can_hold(tmp_path): + """The compute buffer is replicated on every device of a layer split, so a + pooled budget can price a context the smallest card cannot hold; the placement + loop catches that with _every_gpu_holds_reserve and caps to what it does hold. + + A probe comparing pooled footprints only condemns the drafter at that + unattainable context, even though both fit at the context the real placement + must use. The numbers (Auto context, native 8192): the target alone fits on the + big card, the pair fits pooled at 8192, but the 1.5 GB card cannot hold the + 1 GiB pipeline overhead plus its own 8192-token buffer copy, so 5888 is the real + ceiling -- and at 5888 the drafter fits. + """ + mib = 1024**2 + backend, gguf, sidecar = _tight_vram_backend(tmp_path, drafter_gb = 1.0) + backend._read_gguf_metadata = lambda _path: setattr(backend, "_context_length", 8192) + backend._get_gpu_memory = lambda _binary = None: [ + (0, 19_588, 19_588), + (1, 1_546, 1_546), + ] + backend._get_gpu_free_memory = lambda _binary = None: [(0, 19_588), (1, 1_546)] + # Context-linear, so the per-device reserve (and the drafter) shrink with a cap. + backend._compute_buffer_ctx_bytes = lambda n_ctx, *args, **kwargs: n_ctx * 83_886 + backend._estimate_mtp_overhead_bytes = lambda ctx, *args, **kwargs: ctx * 94_371 + # Sanity on the geometry the assertions below rest on (MiB). + assert 1024 + 8192 * 83_886 / mib > 1_546 * 0.97 + assert 1024 + 5888 * 83_886 / mib <= 1_546 * 0.97 + + result = _launch( + backend, + gguf, + dspark_draft_path = str(sidecar), + speculative_type = "auto", + # 0 = Auto context (the branch that caps); the native 8192 above is the target. + n_ctx = 0, + ) + + cmd = result["cmd"] + assert cmd[cmd.index("--model-draft") + 1] == str(sidecar) + assert cmd[cmd.index("--spec-type") + 1] == "draft-dspark" + assert backend.spec_fallback_reason is None diff --git a/studio/frontend/src/features/chat/chat-settings-sheet.tsx b/studio/frontend/src/features/chat/chat-settings-sheet.tsx index 046b26362b..3431a60aab 100644 --- a/studio/frontend/src/features/chat/chat-settings-sheet.tsx +++ b/studio/frontend/src/features/chat/chat-settings-sheet.tsx @@ -426,7 +426,9 @@ function specFallbackMessage({ case "mla_mtp_disabled": return "MTP is disabled by default for this model architecture because it currently runs slower than standard decoding. Choose MTP in the model picker to force it."; case "drafter_no_vram": - return `This model fits in VRAM but its ${drafter} drafter does not, so Auto kept your context length and is running without speculative decoding. Choose ${drafter} in Settings to force it, at a smaller context.`; + // Not "without speculative decoding": the backend puts zero-VRAM ngram-mod + // in the drafter's place where the build has it, so only the drafter is off. + return `This model fits in VRAM but its ${drafter} drafter does not, so Auto kept your context length and turned ${drafter} off for this load. Choose ${drafter} in Settings to force it, at a smaller context.`; case "runtime_error": return `${drafter} could not start for this model on the installed llama.cpp build, so it is running without speculative decoding.`; case "drafter_not_found": From dbccfda9ea9239dccb58f8b3d68a085717709037 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Tue, 11 Aug 2026 14:47:42 +0000 Subject: [PATCH 07/25] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/core/inference/llama_cpp.py | 10 ++++------ 1 file changed, 4 insertions(+), 6 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 212cf9f46a..ddc9b44b53 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -11380,9 +11380,7 @@ def _probe_base(drafter: bool, n: int) -> int: # so the two cannot disagree. Auto only: an explicit context # is honored verbatim, never capped, and overflows to --fit. if not explicit_ctx: - _usable_wo = [ - _gpu_usable(g, _probe_frac(False)) for g in _subset - ] + _usable_wo = [_gpu_usable(g, _probe_frac(False)) for g in _subset] _probe_reserve_at = lambda c, _k = _n: ( (_pipeline_overhead_bytes if _k > 1 else 0) + _cc_bytes(c, _k) // _k @@ -11401,9 +11399,9 @@ def _probe_base(drafter: bool, n: int) -> int: _foot_wo = (_base_wo + _shared) / (1024 * 1024) if _foot_wo > _budget_wo: continue - _foot_w = ( - _probe_base(True, _n) + _shared + _mtp_bytes(_ctx_wo) - ) / (1024 * 1024) + _foot_w = (_probe_base(True, _n) + _shared + _mtp_bytes(_ctx_wo)) / ( + 1024 * 1024 + ) _budget_w = _pool_budget_mib(_subset, _probe_frac(True)) if not _target_fits_somewhere: # The placement this reports on: the first subset that From 23d2b13f8cde4009b4b2bd4c179d7791fd54e96c Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Tue, 11 Aug 2026 19:13:49 +0000 Subject: [PATCH 08/25] Release the drafter reserve on drop, and charge a remote extra-args drafter --- studio/backend/core/inference/llama_cpp.py | 29 ++- studio/backend/routes/inference.py | 97 +++++++++- .../tests/test_chat_load_during_training.py | 168 +++++++++++++++++- .../backend/tests/test_llama_cpp_placement.py | 47 +++++ 4 files changed, 326 insertions(+), 15 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index ddc9b44b53..6bd4c157dc 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -11332,11 +11332,28 @@ def _probe_base(drafter: bool, n: int) -> int: key = lambda g: _gpu_usable(g, _probe_frac(False)), reverse = True, ) + # Same floor as the placement loop's _auto_min_gpus: a tensor + # downgrade raises _layer_min_gpus to keep the request + # multi-GPU, and a one-GPU placement the loop is forbidden to + # choose must not be the one that saves the drafter. + _probe_overhead_mib = _pipeline_overhead_bytes / (1024 * 1024) + _probe_min_gpus = max( + 1, + min( + _layer_min_gpus, + sum( + 1 + for g in _probe_ranked + if _gpu_usable(g, _probe_frac(False)) > _probe_overhead_mib + ) + or 1, + ), + ) _target_fits_somewhere = False _both_fit_somewhere = False _probe_ctx = 0 _probe_need = _probe_have = 0.0 - for _n in range(1, len(_probe_ranked) + 1): + for _n in range(_probe_min_gpus, len(_probe_ranked) + 1): _subset = _probe_ranked[:_n] _cc_n = lambda c, _k = _n: _cc_bytes(c, _k) _base_wo = _probe_base(False, _n) @@ -11418,6 +11435,16 @@ def _probe_base(drafter: bool, n: int) -> int: if _target_fits_somewhere and not _both_fit_somewhere: _spec_dropped_no_vram = True _mtp_will_engage = False + # Clearing the flag alone leaves the reserve in place: the + # eight _mtp_bytes call sites below are unconditional, and + # _fit_context_to_vram invokes any non-None mtp_overhead_fn + # whatever mtp_engaged says. The fit would then still shrink + # the context (or take --fit) for a drafter that no longer + # launches, which is the whole thing this drop prevents. + # _mtp_kv_unsized goes with it, or _flat_mtp_engages below + # would swap the flat fraction in as its replacement. + mtp_overhead_fn = None + _mtp_kv_unsized = False logger.warning( "Speculative decoding disabled for this load: the model " "fits in VRAM at context %d but its drafter does not " diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 0faca419ad..b6fe25c371 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5674,6 +5674,82 @@ def _remote_gguf_companion_bytes( return 0 +# Stand-in for a drafter repo no listing could price. Matches the loader's own +# flat MTP reserve for the case where the drafter's dims are unavailable +# (llama_cpp's _tp_flat_mtp), because it answers the same question: the launch +# will make a drafter resident and nothing here can say how big it is. Zero is +# the one answer that is certainly wrong, since it admits the load. +_REMOTE_DRAFTER_RESERVE_BYTES = 2 * 1024**3 + + +def _split_hf_draft_spec(spec: str) -> tuple[Optional[str], str]: + """``/[:quant]`` -> (repo id, lowercased narrowing hint). + + llama.cpp's common_download_split_repo_tag splits the value on ':', keeps the + tail as the quant tag and then requires the head to be exactly + ``/``, so that is the shape a listing can be asked for. A + trailing ``/.gguf`` is not that shape and llama.cpp rejects it, but it + is a common way to write the flag, and reading the repo out of it prices a + real download instead of falling straight to the flat reserve. Repo None when + nothing repo-shaped is left, which the caller charges as the reserve. + """ + repo, sep, tag = (spec or "").strip().partition(":") + parts = [p for p in repo.split("/") if p] + hint = tag.strip().lower() if sep else "" + if len(parts) > 2 and parts[-1].lower().endswith(".gguf"): + hint = parts[-1].lower() + parts = parts[:2] + if len(parts) != 2: + return None, "" + return "/".join(parts), hint + + +def _remote_drafter_repo_bytes(spec: str, *, hf_token: Optional[str]) -> int: + """Bytes to charge for a drafter named as an HF repo (--spec-draft-hf/-hfd). + + llama-server downloads that repo and loads the drafter out of it exactly as + it does the target model, so it is resident VRAM, but there is no local file + to stat before the load and the target repository's own listing says nothing + about it. Bounded rather than picked, for the reason dflash_budget_bytes + documents and with its arithmetic: which file the fetch lands on is not + knowable from a listing, so the largest WHOLE shard set is the answer a + listing can give, and a split set is charged as the set because llama-server + maps every shard. + + Any failure -- no network, gated repo, malformed id, a repo listing no GGUF + -- falls back to the flat reserve. This guard protects a running training + job, so an unreadable listing must not become a silent charge of zero for a + drafter the launch is certainly going to bring in. + """ + repo, hint = _split_hf_draft_spec(spec) + if not repo: + return _REMOTE_DRAFTER_RESERVE_BYTES + try: + from core.inference.llama_cpp import _gguf_extra_shards + from huggingface_hub import model_info + from utils.models.drafters import dflash_budget_bytes + + info = model_info(repo, token = hf_token, files_metadata = True) + sizes: dict[str, int] = {} + for sibling in info.siblings or []: + name = sibling.rfilename or "" + if not Path(name).name.lower().endswith(".gguf"): + continue + sizes[name] = getattr(sibling, "size", 0) or 0 + if hint: + # The :quant tag (or a named file) is llama.cpp's own narrowing, so + # bounding over the rest of the repo would charge an F16 for a Q4 + # drafter and refuse loads that fit. A tag that matches nothing has + # told us nothing -- repos label quants inconsistently -- and every + # candidate goes back into the bound. + matched = {n: s for n, s in sizes.items() if hint in Path(n).name.lower()} + sizes = matched or sizes + return dflash_budget_bytes(sizes, _gguf_extra_shards) or _REMOTE_DRAFTER_RESERVE_BYTES + except Exception as e: + logger.warning(f"Could not size remote drafter repo {spec}: {e}") + return _REMOTE_DRAFTER_RESERVE_BYTES + + # Upper bound on any current tokenizer, used to rebuild the compute buffer when a # truncated header drops the token array. Above Llama 4 / Gemma 3 (256k), the widest shipping. _ASSUMED_MAX_VOCAB = 262144 @@ -6010,14 +6086,18 @@ def _same_file_key(p: str) -> str: # a remote repo with a local --model-draft still has to price its weights # through the listing, and returning the drafter alone under-estimated a # load by the whole target model. + # A remote one (--spec-draft-hf / -hfd names an HF repo, never a file) is + # charged from its OWN listing, because nothing else charges it: the + # target repository's companion scan only ever sees the target's + # sidecars, so a target that ships none left a multi-GB drafter billed + # nowhere and the guard admitted a load that evicts the training job. _extras_bytes = 0 _extras_draft = _extra_args_mtp_draft_path(llama_extra_args, env = {}) - if ( - _extras_draft - and Path(_extras_draft).is_file() - and _same_file_key(str(_extras_draft)) not in _sized_keys - ): - _extras_bytes = LlamaCppBackend._get_gguf_size_bytes(str(_extras_draft)) + if _extras_draft and Path(_extras_draft).is_file(): + if _same_file_key(str(_extras_draft)) not in _sized_keys: + _extras_bytes = LlamaCppBackend._get_gguf_size_bytes(str(_extras_draft)) + elif _extras_draft: + _extras_bytes = _remote_drafter_repo_bytes(str(_extras_draft), hf_token = hf_token) if total_bytes > 0: return (total_bytes + _extras_bytes) / (1024**3) + _estimate_gguf_kv_gb( @@ -6066,8 +6146,9 @@ def _same_file_key(p: str) -> str: # refusal for bytes that never become resident. dspark_first = _auto_dspark, ) - # Plus the local --model-draft, if the caller named one: the repo - # listing cannot see it, and it is resident next to these weights. + # Plus the caller's own --model-draft / --spec-draft-hf, if they named + # one: this repo's listing cannot see it, local or remote, and it is + # resident next to these weights. total_gb = (main_bytes + companions + _extras_bytes) / (1024**3) # remote dims are unreadable; only the kq mask, linear in ubatch x ctx, can be sized here from core.inference.llama_server_args import parse_ctx_override diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index de019505c5..c4821ad946 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -1596,23 +1596,179 @@ def test_extras_owning_the_spec_type_are_not_charged_the_repo_sidecar(self): # can actually charge, and that charge is gated on Path(...).is_file(). _draft = Path(_tmp.name) / "d.gguf" _draft.write_bytes(b"x" * 512) + # A repo that DOES ship a sidecar, so the assertion is about the resulting + # number and not about which flags were passed: a stub returning 0 whatever + # it is asked would pass even if the suppression stopped working. + _sidecar = 7 * 1024**3 + + def _companions(repo, **kw): + return ( + _sidecar + if (kw["include_mtp"] or kw["include_dspark"] or kw["include_dflash"]) + else 0 + ) + + def _estimate(extras): + with ( + patch.object( + mc, "list_gguf_variants", lambda repo, hf_token = None: ([variant], False) + ), + patch.object(self.route, "_remote_gguf_companion_bytes", _companions), + self._dflash_capable(), + ): + return self.route._estimate_gguf_required_gb( + cfg, speculative_type = "auto", llama_extra_args = extras + ) + + # Same load, minus the private drafter: Auto fetches the repo's sidecar and + # pays for it, which is the charge the suppression has to remove. + baseline = _estimate([]) + self.assertGreater(baseline, _sidecar / (1024**3)) for extras in ( ["--spec-type", "draft-dflash", "--model-draft", str(_draft)], ["--spec-type", "draft-dspark", "--model-draft", str(_draft)], ): + # The repo sidecar gone, their own 512-byte drafter charged in its place. + self.assertAlmostEqual( + _estimate(extras), + baseline - _sidecar / (1024**3) + 512 / (1024**3), + places = 9, + msg = extras, + ) + + def test_a_remote_extras_drafter_is_sized_from_its_own_repository(self): + """--spec-draft-hf/-hfd names a SEPARATE repo, which llama-server downloads + and loads. The target repository's companion scan cannot see it, so a target + that ships no sidecar of its own left the drafter charged nowhere and let the + guard admit a multi-GB overcommit beside a running training job. Bounded by + the largest whole shard set, the only answer a listing can give: which file + the fetch lands on is not knowable, and a split set is resident in full.""" + import utils.models.model_config as mc + from core.inference.llama_cpp import LlamaCppBackend + + cfg = SimpleNamespace( + gguf_file = None, + gguf_mmproj_file = None, + gguf_mtp_file = None, + gguf_dspark_file = None, + gguf_dflash_file = None, + gguf_hf_repo = "org/repo", + gguf_variant = "Q4_K_M", + ) + variant = SimpleNamespace(quant = "Q4_K_M", size_bytes = 1024**3) + siblings = [ + SimpleNamespace(rfilename = "drafter-Q4_K_M-00001-of-00002.gguf", size = 3 * 1024**3 // 2), + SimpleNamespace(rfilename = "drafter-Q4_K_M-00002-of-00002.gguf", size = 3 * 1024**3 // 2), + SimpleNamespace(rfilename = "drafter-Q8_0.gguf", size = 2 * 1024**3), + # Mid-upload: the fetch refuses a short set, so it must not set the bound. + SimpleNamespace(rfilename = "drafter-F16-00001-of-00002.gguf", size = 5 * 1024**3), + SimpleNamespace(rfilename = "notes.md", size = 9 * 1024**3), + ] + + def _estimate(extras): with ( patch.object( mc, "list_gguf_variants", lambda repo, hf_token = None: ([variant], False) ), - patch.object(self.route, "_remote_gguf_companion_bytes", return_value = 0) as comp, - self._dflash_capable(), + # The exact hole: the target repo has no sidecar to be charged for. + patch.object(self.route, "_remote_gguf_companion_bytes", return_value = 0), + patch( + "huggingface_hub.model_info", + return_value = SimpleNamespace(siblings = siblings), + ), + patch.object( + LlamaCppBackend, + "probe_server_capabilities", + classmethod( + lambda cls, binary = None: { + "supports_dspark": True, + "supports_dflash": True, + } + ), + ), ): - self.route._estimate_gguf_required_gb( + return self.route._estimate_gguf_required_gb( cfg, speculative_type = "auto", llama_extra_args = extras ) - self.assertFalse(comp.call_args.kwargs["include_dflash"], extras) - self.assertFalse(comp.call_args.kwargs["include_dspark"], extras) - self.assertFalse(comp.call_args.kwargs["include_mtp"], extras) + + for own, remote in ( + (["--spec-type", "draft-dspark"], ["--spec-draft-hf", "org/drafter"]), + (["--spec-type", "draft-dflash"], ["-hfd", "org/drafter"]), + ): + # Everything else identical, so the difference IS the drafter charge. + self.assertAlmostEqual( + _estimate(own + remote), + _estimate(own) + 3 * 1024**3 / (1024**3), + places = 9, + msg = remote, + ) + # The :quant tag is llama.cpp's own narrowing, so the bound follows it down + # rather than charging the repo's largest family for a small drafter. + self.assertAlmostEqual( + _estimate(["--spec-type", "draft-dflash", "-hfd", "org/drafter:Q8_0"]), + _estimate(["--spec-type", "draft-dflash"]) + 2 * 1024**3 / (1024**3), + places = 9, + ) + + def test_an_unreadable_remote_drafter_repo_still_pays_a_flat_reserve(self): + """No network, a gated repo or a malformed id leaves the drafter unsized, + and this guard protects a running training job: charging zero for a + download the launch is certainly going to make is the one answer that + admits the overcommit.""" + import utils.models.model_config as mc + from core.inference.llama_cpp import LlamaCppBackend + + cfg = SimpleNamespace( + gguf_file = None, + gguf_mmproj_file = None, + gguf_mtp_file = None, + gguf_dspark_file = None, + gguf_dflash_file = None, + gguf_hf_repo = "org/repo", + gguf_variant = "Q4_K_M", + ) + variant = SimpleNamespace(quant = "Q4_K_M", size_bytes = 1024**3) + reserve = self.route._REMOTE_DRAFTER_RESERVE_BYTES + self.assertGreater(reserve, 0) + + def _estimate(extras, listing): + with ( + patch.object( + mc, "list_gguf_variants", lambda repo, hf_token = None: ([variant], False) + ), + patch.object(self.route, "_remote_gguf_companion_bytes", return_value = 0), + patch("huggingface_hub.model_info", **listing), + patch.object( + LlamaCppBackend, + "probe_server_capabilities", + classmethod( + lambda cls, binary = None: { + "supports_dspark": True, + "supports_dflash": True, + } + ), + ), + ): + return self.route._estimate_gguf_required_gb( + cfg, speculative_type = "auto", llama_extra_args = extras + ) + + _raises = {"side_effect": OSError("gated repo")} + _empty = {"return_value": SimpleNamespace(siblings = [])} + base = _estimate(["--spec-type", "draft-dflash"], _raises) + for extras, listing in ( + # Unreadable listing, a repo that lists no GGUF, and an id no listing + # could ever answer for: all of them mean "unsized", not "free". + (["--spec-type", "draft-dflash", "-hfd", "org/drafter"], _raises), + (["--spec-type", "draft-dflash", "-hfd", "org/drafter"], _empty), + (["--spec-type", "draft-dflash", "-hfd", "not-a-repo-id"], _empty), + ): + self.assertAlmostEqual( + _estimate(extras, listing), + base + reserve / (1024**3), + places = 9, + msg = extras, + ) def test_a_remote_extras_drafter_still_pays_the_conservative_charge(self): """--spec-draft-hf names an HF repo, not a file, so _extras_bytes charges diff --git a/studio/backend/tests/test_llama_cpp_placement.py b/studio/backend/tests/test_llama_cpp_placement.py index 04cad582c4..42f05329e2 100644 --- a/studio/backend/tests/test_llama_cpp_placement.py +++ b/studio/backend/tests/test_llama_cpp_placement.py @@ -670,3 +670,50 @@ def test_the_probe_prices_the_drafter_at_a_context_the_weakest_card_can_hold(tmp assert cmd[cmd.index("--model-draft") + 1] == str(sidecar) assert cmd[cmd.index("--spec-type") + 1] == "draft-dspark" assert backend.spec_fallback_reason is None + + +def test_the_drop_actually_releases_the_reserve_the_fit_charges(tmp_path): + """The drop has to reach the fit, not just the launch. + + Every _mtp_bytes site in the fit is unconditional and _fit_context_to_vram + calls any non-None mtp_overhead_fn whatever mtp_engaged says, so clearing + _mtp_will_engage alone still let the planner shrink the context for a drafter + it no longer launches. Deliberately does NOT stub _fit_context_to_vram or the + GPU selectors: the point is the context the real fit arrives at. + """ + gb = 1024**3 + mib = 1024**2 + backend, gguf = _backend(tmp_path, vulkan = False, memory = [(0, 24_576, 24_576)]) + sidecar = tmp_path / "dspark-model-Q8_0.gguf" + sidecar.write_bytes(b"draft") + backend._get_gguf_size_bytes = lambda path: (6 * gb if str(path) == str(sidecar) else 16 * gb) + backend._read_gguf_metadata = lambda _path: setattr(backend, "_context_length", 8192) + backend._can_estimate_kv = lambda: True + # Context-linear, so an unreleased drafter reserve is paid for in context. + backend._estimate_kv_cache_bytes = lambda ctx, *args, **kwargs: int(ctx * 0.5 * mib) + backend._compute_buffer_ctx_bytes = lambda *args, **kwargs: 0 + backend._estimate_compute_buffer_bytes = lambda **kwargs: 1 + backend._mtp_draft_kv_bytes = lambda *args, **kwargs: 0 + backend._estimate_mtp_overhead_bytes = lambda *args, **kwargs: 6 * gb + backend.probe_server_capabilities = lambda _binary = None: { + "supports_dspark": True, + "supports_ngram_mod": True, + "spec_draft_n_max_flag": "--spec-draft-n-max", + } + + result = _launch( + backend, + gguf, + dspark_draft_path = str(sidecar), + speculative_type = "auto", + n_ctx = 0, + ) + + cmd = result["cmd"] + # 16 GB + a 4 GB KV at 8192 clears the 23.3 GB pin budget; + 6 GB does not. + assert "--model-draft" not in cmd + assert backend.spec_fallback_reason == "drafter_no_vram" + # The whole point: native context survives, rather than being cut to pay for + # a drafter that is not launching. + assert cmd[cmd.index("-c") + 1] == "8192" + assert cmd[cmd.index("--fit") + 1] == "off" From 6512d71b0561bf466858666201005fbf229aacce Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Tue, 11 Aug 2026 19:38:08 +0000 Subject: [PATCH 09/25] Cover the largest drafter class in the fallback reserve, and stop double charging a priced override --- studio/backend/routes/inference.py | 20 +++++++++----- .../tests/test_chat_load_during_training.py | 26 ++++++++++++++----- 2 files changed, 33 insertions(+), 13 deletions(-) diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index b6fe25c371..412dbfe33a 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5679,7 +5679,14 @@ def _remote_gguf_companion_bytes( # (llama_cpp's _tp_flat_mtp), because it answers the same question: the launch # will make a drafter resident and nothing here can say how big it is. Zero is # the one answer that is certainly wrong, since it admits the load. -_REMOTE_DRAFTER_RESERVE_BYTES = 2 * 1024**3 +# What an unreadable remote drafter costs the guard. Sized to cover the largest +# drafter class Studio knows of rather than a typical one: a DSpark sidecar is +# about 11 GB (see the --fit note in llama_cpp._emit_dspark), --spec-draft-hf can +# name any repo at all, and this guard protects a running training job, so the +# established direction here is to over-estimate. Only reached when the listing +# cannot be read at all, where llama-server may still open the repo from its +# local HF cache and make every one of those bytes resident. +_REMOTE_DRAFTER_RESERVE_BYTES = 12 * 1024**3 def _split_hf_draft_spec(spec: str) -> tuple[Optional[str], str]: @@ -5955,11 +5962,12 @@ def _estimate_gguf_required_gb( # one would leave a multi-GB resident drafter charged nowhere and let the # guard admit a load that evicts the training job it protects. _extras_own_draft_path = _extra_args_mtp_draft_path(llama_extra_args, env = {}) - _extras_own_drafter = bool( - _extra_args_own_spec - and _extras_own_draft_path - and Path(_extras_own_draft_path).is_file() - ) + # Local file or remote repo alike: both are now charged as _extras_bytes + # below, a local one by stat and a remote one from its own listing (or the + # flat reserve). Charging the target repository's sidecar on top of either + # is the double count that 409s a load which fits. The earlier local-only + # form predates the remote pricing and would now over-charge. + _extras_own_drafter = bool(_extra_args_own_spec and _extras_own_draft_path) _forced_dspark = bool( (_spec_mode == "dspark" or _extra_args_requests_dspark(llama_extra_args, env = {})) and not _extras_own_drafter diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index c4821ad946..a094d8a49d 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -1770,12 +1770,14 @@ def _estimate(extras, listing): msg = extras, ) - def test_a_remote_extras_drafter_still_pays_the_conservative_charge(self): - """--spec-draft-hf names an HF repo, not a file, so _extras_bytes charges - nothing for it while llama-server still downloads and loads it (its - has_dft() only asks whether a draft path was given). Suppressing the - repository sidecar on top would leave the resident drafter charged - nowhere and admit a load that evicts the training job this guard protects.""" + def test_a_priced_remote_extras_drafter_is_not_charged_twice(self): + """--spec-draft-hf names the drafter that actually loads, and it is now + priced from its own listing, so the target repository's sidecar must NOT + be charged as well: _build_speculative_flags returns before Studio emits + that sidecar, so it never becomes resident and billing it 409s a load + that fits. (Before the remote repo was priced this test asserted the + opposite, which was the safe reading while the drafter was charged + nowhere at all.)""" import utils.models.model_config as mc from core.inference.llama_cpp import LlamaCppBackend @@ -1812,7 +1814,17 @@ def test_a_remote_extras_drafter_still_pays_the_conservative_charge(self): self.route._estimate_gguf_required_gb( cfg, speculative_type = "auto", llama_extra_args = extras ) - self.assertTrue(comp.call_args.kwargs[kind], extras) + self.assertFalse(comp.call_args.kwargs[kind], extras) + self.assertFalse(comp.call_args.kwargs["include_mtp"], extras) + + def test_the_unreadable_drafter_reserve_covers_the_largest_drafter_class(self): + """The fallback is only reached when the listing cannot be read, and + llama-server can still open the repo from its local HF cache, so the + number has to cover what it might find. A DSpark sidecar is about 11 GB + (llama_cpp._emit_dspark says so where it warns that --fit skips it), and + --spec-draft-hf can name any repo, so a typical-drafter figure here + underprices the load the guard is protecting a training run from.""" + self.assertGreaterEqual(self.route._REMOTE_DRAFTER_RESERVE_BYTES, 11 * 1024**3) def test_a_partial_dflash_shard_set_is_not_charged(self): """The fetch refuses a family whose encoded shard count is short, so a From 110ae05ea634e0f12711dc8b402e1bcf8218fcf0 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Tue, 11 Aug 2026 20:00:58 +0000 Subject: [PATCH 10/25] Size an unlistable drafter from the cache, and key the CPU exemption on the launched drafter --- studio/backend/core/inference/llama_cpp.py | 54 ++++++++++++------- studio/backend/routes/inference.py | 40 +++++++++++++- .../tests/test_chat_load_during_training.py | 39 ++++++++++++++ .../backend/tests/test_llama_cpp_placement.py | 24 +++++++++ 4 files changed, 136 insertions(+), 21 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 6bd4c157dc..798e73994c 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -11295,7 +11295,15 @@ def _restore_after_tensor_downgrade(): # would release the reserve for a drafter the child still # loads, and the load OOMs. It is also an explicit choice. and not _extra_args_mtp_draft_path(extra_args, env = _spec_env) - and not _draft_cpu_no_embedded + # The drafter this load will launch, not whether an embedded + # head also exists: a separate sidecar wins over one (llama.cpp + # loads the draft model on has_dft()), so with the sidecar + # pinned to CPU there is no GPU drafter reserve to drop. An + # embedded head with no sidecar does sit on GPU, and is probed. + and not ( + _draft_on_cpu + and (mtp_draft_path or _spec_canon in ("dspark", "dflash")) + ) and not tensor_parallel and gpus and effective_ctx > 0 @@ -11396,26 +11404,32 @@ def _probe_base(drafter: bool, n: int) -> int: # / _every_gpu_holds_reserve / _cap_ctx_to_per_device_reserve), # so the two cannot disagree. Auto only: an explicit context # is honored verbatim, never capped, and overflows to --fit. - if not explicit_ctx: - _usable_wo = [_gpu_usable(g, _probe_frac(False)) for g in _subset] - _probe_reserve_at = lambda c, _k = _n: ( - (_pipeline_overhead_bytes if _k > 1 else 0) - + _cc_bytes(c, _k) // _k + _usable_wo = [_gpu_usable(g, _probe_frac(False)) for g in _subset] + _probe_reserve_at = lambda c, _k = _n: ( + (_pipeline_overhead_bytes if _k > 1 else 0) + + _cc_bytes(c, _k) // _k + ) + if not self._every_gpu_holds_reserve( + _usable_wo, _probe_reserve_at(_ctx_wo) + ): + if explicit_ctx: + # Honored verbatim, so there is nothing to cap: this + # subset simply cannot hold it, and _select_gpus_ + # split_aware will say so too by going to --fit. + # Calling it a fit here would drop the drafter and + # then report a pin that never happened. + continue + _ctx_wo = self._cap_ctx_to_per_device_reserve( + _ctx_wo, _usable_wo, _probe_reserve_at ) - if not self._every_gpu_holds_reserve( - _usable_wo, _probe_reserve_at(_ctx_wo) - ): - _ctx_wo = self._cap_ctx_to_per_device_reserve( - _ctx_wo, _usable_wo, _probe_reserve_at - ) - if _ctx_wo <= 0: - continue - # Every pooled term shrinks with the context, so this - # cannot newly fail; re-price rather than lean on it. - _shared = _kv_bytes(_ctx_wo) + _cc_n(_ctx_wo) - _foot_wo = (_base_wo + _shared) / (1024 * 1024) - if _foot_wo > _budget_wo: - continue + if _ctx_wo <= 0: + continue + # Every pooled term shrinks with the context, so this + # cannot newly fail; re-price rather than lean on it. + _shared = _kv_bytes(_ctx_wo) + _cc_n(_ctx_wo) + _foot_wo = (_base_wo + _shared) / (1024 * 1024) + if _foot_wo > _budget_wo: + continue _foot_w = (_probe_base(True, _n) + _shared + _mtp_bytes(_ctx_wo)) / ( 1024 * 1024 ) diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 412dbfe33a..f999e7ceac 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5754,7 +5754,45 @@ def _remote_drafter_repo_bytes(spec: str, *, hf_token: Optional[str]) -> int: return dflash_budget_bytes(sizes, _gguf_extra_shards) or _REMOTE_DRAFTER_RESERVE_BYTES except Exception as e: logger.warning(f"Could not size remote drafter repo {spec}: {e}") - return _REMOTE_DRAFTER_RESERVE_BYTES + # Unreadable listings are the case where the repo is already in the local HF + # cache: that is what lets llama-server open it with the Hub unreachable, and + # it is also what makes the flat reserve dangerous, since --spec-draft-hf takes + # any repo and an ordinary 30 GB GGUF is a legal value. Measure the cache. + cached = _cached_repo_gguf_bytes(repo) + if cached: + return cached + # Neither listable nor cached: llama-server would have to download it over the + # same Hub that just refused us, so the reserve is a cushion for a drafter that + # most likely never arrives, not a bound on one that has. + return _REMOTE_DRAFTER_RESERVE_BYTES + + +def _cached_repo_gguf_bytes(repo: str) -> int: + """Largest whole GGUF shard set already on disk for ``repo``, else 0. + + Same bound as the listing path, taken from the local Hugging Face cache, so a + drafter llama-server can open offline is charged at its real size rather than + a class-based guess. + """ + try: + from huggingface_hub import scan_cache_dir + + from core.inference.llama_cpp import _gguf_extra_shards + from utils.models.drafters import dflash_budget_bytes + + sizes: dict[str, int] = {} + for cached_repo in scan_cache_dir().repos: + if (cached_repo.repo_id or "").lower() != repo.lower(): + continue + for revision in cached_repo.revisions: + for f in revision.files: + name = str(f.file_name) + if name.lower().endswith(".gguf"): + sizes[name] = max(sizes.get(name, 0), int(f.size_on_disk or 0)) + return dflash_budget_bytes(sizes, _gguf_extra_shards) + except Exception as e: + logger.warning(f"Could not measure the cached drafter repo {repo}: {e}") + return 0 # Upper bound on any current tokenizer, used to rebuild the compute buffer when a diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index a094d8a49d..b447f177ae 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -1826,6 +1826,45 @@ def test_the_unreadable_drafter_reserve_covers_the_largest_drafter_class(self): underprices the load the guard is protecting a training run from.""" self.assertGreaterEqual(self.route._REMOTE_DRAFTER_RESERVE_BYTES, 11 * 1024**3) + def test_an_unlistable_remote_drafter_is_measured_from_the_local_cache(self): + """An unreadable listing is exactly the case where the repo is already + cached, which is what lets llama-server open it offline, and --spec-draft-hf + takes any repo, so a class-based constant can undercount a 30 GB drafter by + a lot. Measure what is on disk instead, with the same whole-shard-set bound + the listing path uses.""" + cached = SimpleNamespace( + repo_id = "org/drafter", + revisions = [ + SimpleNamespace( + files = [ + SimpleNamespace(file_name = "big-00001-of-00002.gguf", size_on_disk = 16 * 1024**3), + SimpleNamespace(file_name = "big-00002-of-00002.gguf", size_on_disk = 14 * 1024**3), + SimpleNamespace(file_name = "notes.txt", size_on_disk = 10), + ] + ) + ], + ) + with ( + patch("huggingface_hub.model_info", side_effect = OSError("no network")), + patch( + "huggingface_hub.scan_cache_dir", + return_value = SimpleNamespace(repos = [cached]), + ), + ): + charged = self.route._remote_drafter_repo_bytes("org/drafter", hf_token = None) + # The whole 30 GB set that is actually resident, not the flat reserve. + self.assertEqual(charged, 30 * 1024**3) + + def test_a_remote_drafter_that_is_neither_listable_nor_cached_pays_the_reserve(self): + """Nothing to measure and nothing to download over the Hub that just + refused the listing, so the reserve is a cushion rather than a bound.""" + with ( + patch("huggingface_hub.model_info", side_effect = OSError("no network")), + patch("huggingface_hub.scan_cache_dir", return_value = SimpleNamespace(repos = [])), + ): + charged = self.route._remote_drafter_repo_bytes("org/drafter", hf_token = None) + self.assertEqual(charged, self.route._REMOTE_DRAFTER_RESERVE_BYTES) + def test_a_partial_dflash_shard_set_is_not_charged(self): """The fetch refuses a family whose encoded shard count is short, so a listing caught mid-publication must not be billed for its listed half: diff --git a/studio/backend/tests/test_llama_cpp_placement.py b/studio/backend/tests/test_llama_cpp_placement.py index 42f05329e2..c44ab5c54f 100644 --- a/studio/backend/tests/test_llama_cpp_placement.py +++ b/studio/backend/tests/test_llama_cpp_placement.py @@ -717,3 +717,27 @@ def test_the_drop_actually_releases_the_reserve_the_fit_charges(tmp_path): # a drafter that is not launching. assert cmd[cmd.index("-c") + 1] == "8192" assert cmd[cmd.index("--fit") + 1] == "off" + + +def test_a_cpu_offloaded_sidecar_is_not_probed_because_a_head_also_exists(tmp_path): + """The drafter that launches decides, not one that merely exists. + + llama.cpp loads the draft model on has_dft(), so a separate sidecar wins over + an embedded head; pinned to CPU it takes no GPU reserve, and there is nothing + for the shortfall probe to drop. Keying the exemption on "no embedded head" + dropped a sidecar that was never on the GPU in the first place. + """ + backend, gguf, sidecar = _tight_vram_backend(tmp_path, drafter_gb = 12.0) + backend._read_gguf_metadata = lambda _path: setattr(backend, "_nextn_predict_layers", 1) + + result = _launch( + backend, + gguf, + dspark_draft_path = str(sidecar), + speculative_type = "auto", + n_ctx = 8192, + extra_args = ["--spec-draft-ngl", "0"], + ) + + assert backend.spec_fallback_reason != "drafter_no_vram" + assert "--model-draft" in result["cmd"] From 0c4e21ab4cc7fe047e0a58c2aa3c962023c8b77c Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Tue, 11 Aug 2026 20:03:28 +0000 Subject: [PATCH 11/25] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/core/inference/llama_cpp.py | 3 +-- studio/backend/tests/test_chat_load_during_training.py | 8 ++++++-- 2 files changed, 7 insertions(+), 4 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 798e73994c..e56e1b83a1 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -11406,8 +11406,7 @@ def _probe_base(drafter: bool, n: int) -> int: # is honored verbatim, never capped, and overflows to --fit. _usable_wo = [_gpu_usable(g, _probe_frac(False)) for g in _subset] _probe_reserve_at = lambda c, _k = _n: ( - (_pipeline_overhead_bytes if _k > 1 else 0) - + _cc_bytes(c, _k) // _k + (_pipeline_overhead_bytes if _k > 1 else 0) + _cc_bytes(c, _k) // _k ) if not self._every_gpu_holds_reserve( _usable_wo, _probe_reserve_at(_ctx_wo) diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index b447f177ae..243ff1acbe 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -1837,8 +1837,12 @@ def test_an_unlistable_remote_drafter_is_measured_from_the_local_cache(self): revisions = [ SimpleNamespace( files = [ - SimpleNamespace(file_name = "big-00001-of-00002.gguf", size_on_disk = 16 * 1024**3), - SimpleNamespace(file_name = "big-00002-of-00002.gguf", size_on_disk = 14 * 1024**3), + SimpleNamespace( + file_name = "big-00001-of-00002.gguf", size_on_disk = 16 * 1024**3 + ), + SimpleNamespace( + file_name = "big-00002-of-00002.gguf", size_on_disk = 14 * 1024**3 + ), SimpleNamespace(file_name = "notes.txt", size_on_disk = 10), ] ) From 7c7d2d416b39b4db8d852a7b89872ba3a6727abf Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Tue, 11 Aug 2026 20:26:04 +0000 Subject: [PATCH 12/25] Narrow the cached drafter by quant, and keep a CPU drafter out of the VRAM reserve --- studio/backend/core/inference/llama_cpp.py | 21 +++--- studio/backend/routes/inference.py | 18 ++++-- .../tests/test_chat_load_during_training.py | 64 +++++++++++++++++++ .../backend/tests/test_llama_cpp_placement.py | 31 +++++++++ studio/backend/tests/test_mtp_vram_budget.py | 13 ++-- 5 files changed, 128 insertions(+), 19 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index e56e1b83a1..3fc76450bd 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -11156,7 +11156,16 @@ def _cc_split_extra(ctx: int) -> int: # llama.cpp split by free VRAM). tp_tensor_split: Optional[list[int]] = None explicit_ctx = requested_ctx > 0 - _draft_cpu_no_embedded = _draft_on_cpu and not self._nextn_predict_layers + # CPU-offloaded and the drafter that will launch is the separate + # one, so nothing speculative sits on the GPU. A separate sidecar + # wins over an embedded head (llama.cpp loads the draft model on + # has_dft()), so an unused head must not keep the reserve alive; + # a head with no sidecar is still GPU-resident and still reserved. + _draft_cpu_no_embedded = _draft_on_cpu and ( + bool(mtp_draft_path) + or _spec_canon in ("dspark", "dflash") + or not self._nextn_predict_layers + ) # The two tensor -> layer downgrades that depend on nothing the # drafter probe below decides run here, BEFORE it: the probe is @@ -11295,15 +11304,7 @@ def _restore_after_tensor_downgrade(): # would release the reserve for a drafter the child still # loads, and the load OOMs. It is also an explicit choice. and not _extra_args_mtp_draft_path(extra_args, env = _spec_env) - # The drafter this load will launch, not whether an embedded - # head also exists: a separate sidecar wins over one (llama.cpp - # loads the draft model on has_dft()), so with the sidecar - # pinned to CPU there is no GPU drafter reserve to drop. An - # embedded head with no sidecar does sit on GPU, and is probed. - and not ( - _draft_on_cpu - and (mtp_draft_path or _spec_canon in ("dspark", "dflash")) - ) + and not _draft_cpu_no_embedded and not tensor_parallel and gpus and effective_ctx > 0 diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index f999e7ceac..1902abb1aa 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5758,7 +5758,7 @@ def _remote_drafter_repo_bytes(spec: str, *, hf_token: Optional[str]) -> int: # cache: that is what lets llama-server open it with the Hub unreachable, and # it is also what makes the flat reserve dangerous, since --spec-draft-hf takes # any repo and an ordinary 30 GB GGUF is a legal value. Measure the cache. - cached = _cached_repo_gguf_bytes(repo) + cached = _cached_repo_gguf_bytes(repo, hint) if cached: return cached # Neither listable nor cached: llama-server would have to download it over the @@ -5767,12 +5767,13 @@ def _remote_drafter_repo_bytes(spec: str, *, hf_token: Optional[str]) -> int: return _REMOTE_DRAFTER_RESERVE_BYTES -def _cached_repo_gguf_bytes(repo: str) -> int: +def _cached_repo_gguf_bytes(repo: str, hint: str = "") -> int: """Largest whole GGUF shard set already on disk for ``repo``, else 0. Same bound as the listing path, taken from the local Hugging Face cache, so a drafter llama-server can open offline is charged at its real size rather than - a class-based guess. + a class-based guess. ``hint`` narrows exactly as it does there: a repo holding + several quants would otherwise be charged its F16 for a :Q4_K_M request. """ try: from huggingface_hub import scan_cache_dir @@ -5789,6 +5790,8 @@ def _cached_repo_gguf_bytes(repo: str) -> int: name = str(f.file_name) if name.lower().endswith(".gguf"): sizes[name] = max(sizes.get(name, 0), int(f.size_on_disk or 0)) + if hint: + sizes = {n: b for n, b in sizes.items() if hint in n.lower()} or sizes return dflash_budget_bytes(sizes, _gguf_extra_shards) except Exception as e: logger.warning(f"Could not measure the cached drafter repo {repo}: {e}") @@ -5977,6 +5980,7 @@ def _estimate_gguf_required_gb( try: from core.inference.llama_cpp import ( _canonicalize_spec_mode, + _extra_args_draft_offloaded_to_cpu, _extra_args_mtp_draft_path, _extra_args_requests_dflash, _extra_args_requests_dspark, @@ -6139,7 +6143,13 @@ def _same_file_key(p: str) -> str: # nowhere and the guard admitted a load that evicts the training job. _extras_bytes = 0 _extras_draft = _extra_args_mtp_draft_path(llama_extra_args, env = {}) - if _extras_draft and Path(_extras_draft).is_file(): + # -ngld 0 / --spec-draft-device cpu keeps the drafter in host memory, so it + # competes for RAM, not for the training job's VRAM. Charging it here is not + # the safe over-estimate the rest of this guard makes; it is simply the wrong + # resource, and a large drafter then 409s a load that takes no VRAM at all. + if _extra_args_draft_offloaded_to_cpu(llama_extra_args, env = os.environ): + _extras_bytes = 0 + elif _extras_draft and Path(_extras_draft).is_file(): if _same_file_key(str(_extras_draft)) not in _sized_keys: _extras_bytes = LlamaCppBackend._get_gguf_size_bytes(str(_extras_draft)) elif _extras_draft: diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index 243ff1acbe..0d4f77ca1a 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -1869,6 +1869,70 @@ def test_a_remote_drafter_that_is_neither_listable_nor_cached_pays_the_reserve(s charged = self.route._remote_drafter_repo_bytes("org/drafter", hf_token = None) self.assertEqual(charged, self.route._REMOTE_DRAFTER_RESERVE_BYTES) + def test_a_cached_drafter_is_narrowed_by_the_quant_tag(self): + """llama.cpp's :quant is its own narrowing, so a repo holding several + quants must not be charged its F16 for a Q4_K_M request. Same rule the + listing path applies, and the two have to agree.""" + cached = SimpleNamespace( + repo_id = "org/drafter", + revisions = [ + SimpleNamespace( + files = [ + SimpleNamespace( + file_name = "drafter-F16.gguf", size_on_disk = 20 * 1024**3 + ), + SimpleNamespace( + file_name = "drafter-Q4_K_M.gguf", size_on_disk = 3 * 1024**3 + ), + ] + ) + ], + ) + with ( + patch("huggingface_hub.model_info", side_effect = OSError("no network")), + patch( + "huggingface_hub.scan_cache_dir", + return_value = SimpleNamespace(repos = [cached]), + ), + ): + charged = self.route._remote_drafter_repo_bytes( + "org/drafter:Q4_K_M", hf_token = None + ) + self.assertEqual(charged, 3 * 1024**3) + + def test_a_cpu_offloaded_extras_drafter_is_not_charged_vram(self): + """-ngld 0 keeps the drafter in host memory. Charging it against the + training job's VRAM is not a conservative estimate, it is the wrong + resource, and it 409s a load that takes no VRAM for the drafter at all.""" + import utils.models.model_config as mc + + cfg = SimpleNamespace( + gguf_file = None, + gguf_mmproj_file = None, + gguf_mtp_file = None, + gguf_dspark_file = None, + gguf_dflash_file = None, + gguf_hf_repo = "org/repo", + gguf_variant = "Q4_K_M", + ) + variant = SimpleNamespace(quant = "Q4_K_M", size_bytes = 1024**3) + extras = ["--spec-type", "draft-dspark", "--spec-draft-hf", "org/drafter"] + with ( + patch.object(mc, "list_gguf_variants", lambda repo, hf_token = None: ([variant], False)), + patch.object(self.route, "_remote_gguf_companion_bytes", return_value = 0), + patch.object(self.route, "_remote_drafter_repo_bytes", return_value = 8 * 1024**3), + patch.object(self.route, "_estimate_gguf_kv_gb", return_value = 0.0), + ): + on_gpu = self.route._estimate_gguf_required_gb( + cfg, speculative_type = "auto", llama_extra_args = extras + ) + on_cpu = self.route._estimate_gguf_required_gb( + cfg, + speculative_type = "auto", + llama_extra_args = [*extras, "--spec-draft-ngl", "0"], + ) + self.assertAlmostEqual(on_gpu - on_cpu, 8.0, places = 6) + def test_a_partial_dflash_shard_set_is_not_charged(self): """The fetch refuses a family whose encoded shard count is short, so a listing caught mid-publication must not be billed for its listed half: diff --git a/studio/backend/tests/test_llama_cpp_placement.py b/studio/backend/tests/test_llama_cpp_placement.py index c44ab5c54f..28011c7582 100644 --- a/studio/backend/tests/test_llama_cpp_placement.py +++ b/studio/backend/tests/test_llama_cpp_placement.py @@ -741,3 +741,34 @@ def test_a_cpu_offloaded_sidecar_is_not_probed_because_a_head_also_exists(tmp_pa assert backend.spec_fallback_reason != "drafter_no_vram" assert "--model-draft" in result["cmd"] + + +def test_a_cpu_offloaded_sidecar_reserves_no_gpu_despite_an_embedded_head(tmp_path): + """The exemption has to reach the reserve, not just the probe. + + _mtp_reserves_gpu kept the flat fraction and draft-compute reserve alive for + an embedded head the launch never uses, so the context still shrank for GPU + memory nothing allocates. One definition now serves both. + """ + backend, gguf, sidecar = _tight_vram_backend(tmp_path, drafter_gb = 12.0) + def _meta(_path): + backend._nextn_predict_layers = 1 + backend._context_length = 8192 + + backend._read_gguf_metadata = _meta + reserved = [] + backend._fit_context_to_vram = lambda requested, *a, **k: ( + reserved.append(k.get("mtp_engaged")) or requested + ) + + _launch( + backend, + gguf, + dspark_draft_path = str(sidecar), + speculative_type = "auto", + n_ctx = 0, + extra_args = ["--spec-draft-ngl", "0"], + ) + + assert reserved, "the fit never ran" + assert not any(reserved), f"mtp_engaged should be False throughout, got {reserved}" diff --git a/studio/backend/tests/test_mtp_vram_budget.py b/studio/backend/tests/test_mtp_vram_budget.py index 79e997f4b2..ce4de29172 100644 --- a/studio/backend/tests/test_mtp_vram_budget.py +++ b/studio/backend/tests/test_mtp_vram_budget.py @@ -598,17 +598,20 @@ def test_load_model_drafter_budget_precedence(self): def test_load_model_drops_cpu_offloaded_drafter_from_budget(self): # A SEPARATE drafter offloaded to CPU (--spec-draft-ngl 0 / # --spec-draft-device none) consumes no GPU, so it must be dropped from the - # budget and get no flat reserve (Finding F2). But an embedded head is on - # GPU regardless of those draft-only flags, so the flat reserve is only - # suppressed when there is no embedded head (Finding G5). + # budget and get no flat reserve (Finding F2). An embedded head is on GPU + # regardless of those draft-only flags (Finding G5) -- but only when it is + # the head that runs: llama.cpp loads the draft model on has_dft(), so a + # separate sidecar wins over one, and an unused head must not keep the + # reserve alive. Hence "no embedded head OR a separate drafter was chosen". compact = "".join(inspect.getsource(LlamaCppBackend.load_model).split()) # env-aware: also honors the inherited LLAMA_ARG_N_GPU_LAYERS_DRAFT. assert ( "_draft_on_cpu=_extra_args_draft_offloaded_to_cpu(extra_args,env=os.environ)" in compact ) assert "if_draft_on_cpu:_mtp_draft_for_budget=None" in compact - # flat reserve suppressed only for a CPU drafter with no embedded head - assert "_draft_cpu_no_embedded=_draft_on_cpuandnotself._nextn_predict_layers" in compact + # flat reserve suppressed for a CPU drafter that is the one being launched + assert "_draft_cpu_no_embedded=_draft_on_cpuand(" in compact + assert "ornotself._nextn_predict_layers)" in compact assert "not_draft_cpu_no_embedded" in compact def test_load_model_keeps_flat_reserve_for_unsized_draft_kv(self): From 2d51f6857142c59993cb35b2fa902a0222f394f4 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Tue, 11 Aug 2026 20:29:08 +0000 Subject: [PATCH 13/25] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- .../backend/tests/test_chat_load_during_training.py | 12 +++--------- studio/backend/tests/test_llama_cpp_placement.py | 1 + 2 files changed, 4 insertions(+), 9 deletions(-) diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index 0d4f77ca1a..9f5ce59ebc 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -1878,12 +1878,8 @@ def test_a_cached_drafter_is_narrowed_by_the_quant_tag(self): revisions = [ SimpleNamespace( files = [ - SimpleNamespace( - file_name = "drafter-F16.gguf", size_on_disk = 20 * 1024**3 - ), - SimpleNamespace( - file_name = "drafter-Q4_K_M.gguf", size_on_disk = 3 * 1024**3 - ), + SimpleNamespace(file_name = "drafter-F16.gguf", size_on_disk = 20 * 1024**3), + SimpleNamespace(file_name = "drafter-Q4_K_M.gguf", size_on_disk = 3 * 1024**3), ] ) ], @@ -1895,9 +1891,7 @@ def test_a_cached_drafter_is_narrowed_by_the_quant_tag(self): return_value = SimpleNamespace(repos = [cached]), ), ): - charged = self.route._remote_drafter_repo_bytes( - "org/drafter:Q4_K_M", hf_token = None - ) + charged = self.route._remote_drafter_repo_bytes("org/drafter:Q4_K_M", hf_token = None) self.assertEqual(charged, 3 * 1024**3) def test_a_cpu_offloaded_extras_drafter_is_not_charged_vram(self): diff --git a/studio/backend/tests/test_llama_cpp_placement.py b/studio/backend/tests/test_llama_cpp_placement.py index 28011c7582..5505eb0bd7 100644 --- a/studio/backend/tests/test_llama_cpp_placement.py +++ b/studio/backend/tests/test_llama_cpp_placement.py @@ -751,6 +751,7 @@ def test_a_cpu_offloaded_sidecar_reserves_no_gpu_despite_an_embedded_head(tmp_pa memory nothing allocates. One definition now serves both. """ backend, gguf, sidecar = _tight_vram_backend(tmp_path, drafter_gb = 12.0) + def _meta(_path): backend._nextn_predict_layers = 1 backend._context_length = 8192 From 022bd7e17a000c914d2afcdcfdd7ea1c794003fb Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Tue, 11 Aug 2026 20:53:29 +0000 Subject: [PATCH 14/25] Decide on the placement the loop takes, and read one cached revision --- studio/backend/core/inference/llama_cpp.py | 46 +++++++++++++++++-- studio/backend/routes/inference.py | 25 ++++++++-- .../tests/test_chat_load_during_training.py | 15 +++++- .../backend/tests/test_llama_cpp_placement.py | 44 ++++++++++++++++++ 4 files changed, 118 insertions(+), 12 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 3fc76450bd..c84b3ff197 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -10987,6 +10987,12 @@ def _pool_budget_mib(subset, frac): _mtp_draft_for_budget = ( _cli_draft_for_budget or _studio_draft_for_budget or _env_draft_for_budget ) + # The same three sources, kept before the CPU nulling below, as the + # answer to "will a SEPARATE drafter be emitted at all". Not + # mtp_draft_path: extras owning --spec-type return from + # _build_speculative_flags before Studio's resolved sidecar becomes + # --model-draft, and _studio_draft_for_budget already encodes that. + _separate_draft_launches = bool(_mtp_draft_for_budget) # Drafter offloaded to CPU keeps its weights+KV off the GPU, so # drop it from the budget (an embedded head stays in the model). # Consult the env too: the child honors LLAMA_ARG_N_GPU_LAYERS_DRAFT. @@ -11162,9 +11168,7 @@ def _cc_split_extra(ctx: int) -> int: # has_dft()), so an unused head must not keep the reserve alive; # a head with no sidecar is still GPU-resident and still reserved. _draft_cpu_no_embedded = _draft_on_cpu and ( - bool(mtp_draft_path) - or _spec_canon in ("dspark", "dflash") - or not self._nextn_predict_layers + _separate_draft_launches or not self._nextn_predict_layers ) # The two tensor -> layer downgrades that depend on nothing the @@ -11435,8 +11439,6 @@ def _probe_base(drafter: bool, n: int) -> int: ) _budget_w = _pool_budget_mib(_subset, _probe_frac(True)) if not _target_fits_somewhere: - # The placement this reports on: the first subset that - # holds the target, which is the one the loop would pick. _target_fits_somewhere = True _probe_ctx, _probe_need, _probe_have = ( _ctx_wo, @@ -11446,6 +11448,40 @@ def _probe_base(drafter: bool, n: int) -> int: if _foot_w <= _budget_w: _both_fit_somewhere = True break + # Both do not fit at the target's own context. The placement + # loop does not simply move on: it re-caps the context WITH + # the drafter charged and accepts this subset at whatever + # that leaves. So if a smaller context would hold both here, + # this is the placement the load takes and the drafter is + # paid for in context, which is the trade being refused. It + # only moves to a larger subset when no context fits both. + _ctx_w = ( + 0 + if explicit_ctx + else self._fit_context_to_vram( + _ctx_wo, + _budget_w, + _probe_base(True, _n), + cache_type_kv, + swa_full = swa_full, + n_parallel = n_parallel, + kv_unified = planned_kv_unified, + n_ubatch = _effective_ubatch, + flash_attn = planned_flash_attn, + mtp_engaged = True, + mtp_overhead_fn = mtp_overhead_fn, + compute_ctx_bytes_fn = _cc_n, + budget_frac = 1.0, + total_mib = None, + ) + ) + if _ctx_w > 0 and ( + _probe_base(True, _n) + + _kv_bytes(_ctx_w) + + _cc_n(_ctx_w) + + _mtp_bytes(_ctx_w) + ) / (1024 * 1024) <= _budget_w: + break if _target_fits_somewhere and not _both_fit_somewhere: _spec_dropped_no_vram = True _mtp_will_engage = False diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 1902abb1aa..76b1f25b99 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5785,11 +5785,26 @@ def _cached_repo_gguf_bytes(repo: str, hint: str = "") -> int: for cached_repo in scan_cache_dir().repos: if (cached_repo.repo_id or "").lower() != repo.lower(): continue - for revision in cached_repo.revisions: - for f in revision.files: - name = str(f.file_name) - if name.lower().endswith(".gguf"): - sizes[name] = max(sizes.get(name, 0), int(f.size_on_disk or 0)) + # One revision, not every snapshot on disk. llama-server resolves the + # cached ref it was asked for (main by default), so merging stale + # snapshots -- and taking the largest historical size per filename -- + # charges a quant that was replaced months ago and 409s a load the + # current drafter fits inside. Prefer the ref, else the newest. + revisions = list(cached_repo.revisions) + chosen = next( + ( + r + for r in revisions + if "main" in {str(x) for x in (getattr(r, "refs", None) or ())} + ), + None, + ) or max( + revisions, key = lambda r: getattr(r, "last_modified", 0) or 0, default = None + ) + for f in getattr(chosen, "files", ()) or (): + name = str(f.file_name) + if name.lower().endswith(".gguf"): + sizes[name] = max(sizes.get(name, 0), int(f.size_on_disk or 0)) if hint: sizes = {n: b for n, b in sizes.items() if hint in n.lower()} or sizes return dflash_budget_bytes(sizes, _gguf_extra_shards) diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index 9f5ce59ebc..6494a3c3ab 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -1836,6 +1836,8 @@ def test_an_unlistable_remote_drafter_is_measured_from_the_local_cache(self): repo_id = "org/drafter", revisions = [ SimpleNamespace( + refs = {"main"}, + last_modified = 2.0, files = [ SimpleNamespace( file_name = "big-00001-of-00002.gguf", size_on_disk = 16 * 1024**3 @@ -1844,8 +1846,17 @@ def test_an_unlistable_remote_drafter_is_measured_from_the_local_cache(self): file_name = "big-00002-of-00002.gguf", size_on_disk = 14 * 1024**3 ), SimpleNamespace(file_name = "notes.txt", size_on_disk = 10), - ] - ) + ], + ), + # A stale snapshot still on disk. llama-server resolves the cached + # ref it was asked for, so this relic must not become the bound. + SimpleNamespace( + refs = set(), + last_modified = 1.0, + files = [ + SimpleNamespace(file_name = "old-F16.gguf", size_on_disk = 60 * 1024**3) + ], + ), ], ) with ( diff --git a/studio/backend/tests/test_llama_cpp_placement.py b/studio/backend/tests/test_llama_cpp_placement.py index 5505eb0bd7..3e1c20b7cc 100644 --- a/studio/backend/tests/test_llama_cpp_placement.py +++ b/studio/backend/tests/test_llama_cpp_placement.py @@ -773,3 +773,47 @@ def _meta(_path): assert reserved, "the fit never ran" assert not any(reserved), f"mtp_engaged should be False throughout, got {reserved}" + + +def test_a_subset_that_can_shrink_to_hold_both_is_where_the_decision_lands(tmp_path): + """The placement loop does not walk past a subset that fails with the drafter. + + It re-caps the context WITH the drafter charged and accepts that subset at + whatever is left, so a smaller context holding both here IS the placement the + load takes, and the drafter gets paid for in context. Believing a later, + larger subset would rescue it keeps the drafter and shrinks the context, which + is the trade this exists to refuse. + """ + gb = 1024**3 + backend, gguf = _backend( + tmp_path, vulkan = False, memory = [(0, 24_576, 24_576), (1, 24_576, 24_576)] + ) + sidecar = tmp_path / "dspark-model-Q8_0.gguf" + sidecar.write_bytes(b"draft") + backend._get_gguf_size_bytes = lambda path: (8 * gb if str(path) == str(sidecar) else 16 * gb) + backend._read_gguf_metadata = lambda _path: setattr(backend, "_context_length", 8192) + backend._can_estimate_kv = lambda: True + # Context-linear throughout, so GPU0 alone can shrink its way to holding both. + backend._estimate_kv_cache_bytes = lambda ctx, *a, **k: int(ctx * 0.5 * 1024**2) + backend._compute_buffer_ctx_bytes = lambda *a, **k: 0 + backend._estimate_compute_buffer_bytes = lambda **k: 1 + backend._mtp_draft_kv_bytes = lambda *a, **k: 0 + backend._estimate_mtp_overhead_bytes = lambda ctx, *a, **k: int(ctx * 0.75 * 1024**2) + backend.probe_server_capabilities = lambda _binary = None: { + "supports_dspark": True, + "supports_ngram_mod": True, + "spec_draft_n_max_flag": "--spec-draft-n-max", + } + + result = _launch( + backend, + gguf, + dspark_draft_path = str(sidecar), + speculative_type = "auto", + n_ctx = 0, + ) + + cmd = result["cmd"] + assert "--model-draft" not in cmd + assert backend.spec_fallback_reason == "drafter_no_vram" + assert cmd[cmd.index("-c") + 1] == "8192" From 56b6c3867d0a46a355220e2f9e62dbd985b3467b Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Tue, 11 Aug 2026 20:55:12 +0000 Subject: [PATCH 15/25] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/core/inference/llama_cpp.py | 17 +++++++++++------ studio/backend/routes/inference.py | 4 +--- .../tests/test_chat_load_during_training.py | 4 +--- 3 files changed, 13 insertions(+), 12 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index c84b3ff197..791c750f90 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -11475,12 +11475,17 @@ def _probe_base(drafter: bool, n: int) -> int: total_mib = None, ) ) - if _ctx_w > 0 and ( - _probe_base(True, _n) - + _kv_bytes(_ctx_w) - + _cc_n(_ctx_w) - + _mtp_bytes(_ctx_w) - ) / (1024 * 1024) <= _budget_w: + if ( + _ctx_w > 0 + and ( + _probe_base(True, _n) + + _kv_bytes(_ctx_w) + + _cc_n(_ctx_w) + + _mtp_bytes(_ctx_w) + ) + / (1024 * 1024) + <= _budget_w + ): break if _target_fits_somewhere and not _both_fit_somewhere: _spec_dropped_no_vram = True diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 76b1f25b99..3947e15415 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5798,9 +5798,7 @@ def _cached_repo_gguf_bytes(repo: str, hint: str = "") -> int: if "main" in {str(x) for x in (getattr(r, "refs", None) or ())} ), None, - ) or max( - revisions, key = lambda r: getattr(r, "last_modified", 0) or 0, default = None - ) + ) or max(revisions, key = lambda r: getattr(r, "last_modified", 0) or 0, default = None) for f in getattr(chosen, "files", ()) or (): name = str(f.file_name) if name.lower().endswith(".gguf"): diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index 6494a3c3ab..d16a11089a 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -1853,9 +1853,7 @@ def test_an_unlistable_remote_drafter_is_measured_from_the_local_cache(self): SimpleNamespace( refs = set(), last_modified = 1.0, - files = [ - SimpleNamespace(file_name = "old-F16.gguf", size_on_disk = 60 * 1024**3) - ], + files = [SimpleNamespace(file_name = "old-F16.gguf", size_on_disk = 60 * 1024**3)], ), ], ) From 65b3a31cf18775d39d36d03d9aae25a70c5f8d2f Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Tue, 11 Aug 2026 21:19:29 +0000 Subject: [PATCH 16/25] Let an extras draft path and a CPU pin displace the repository sidecar charge --- studio/backend/routes/inference.py | 38 ++++++--- .../tests/test_chat_load_during_training.py | 80 ++++++++++++++++++- 2 files changed, 105 insertions(+), 13 deletions(-) diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 3947e15415..19c1adc6d5 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5749,7 +5749,10 @@ def _remote_drafter_repo_bytes(spec: str, *, hf_token: Optional[str]) -> int: # drafter and refuse loads that fit. A tag that matches nothing has # told us nothing -- repos label quants inconsistently -- and every # candidate goes back into the bound. - matched = {n: s for n, s in sizes.items() if hint in Path(n).name.lower()} + # Full relative name, not the basename: the quant-subdirectory layout + # (Q4_K_M/model.gguf) is supported everywhere else, and matching only + # the basename restores every quant and charges the repo's F16. + matched = {n: s for n, s in sizes.items() if hint in n.lower()} sizes = matched or sizes return dflash_budget_bytes(sizes, _gguf_extra_shards) or _REMOTE_DRAFTER_RESERVE_BYTES except Exception as e: @@ -6017,19 +6020,30 @@ def _estimate_gguf_required_gb( # one would leave a multi-GB resident drafter charged nowhere and let the # guard admit a load that evicts the training job it protects. _extras_own_draft_path = _extra_args_mtp_draft_path(llama_extra_args, env = {}) - # Local file or remote repo alike: both are now charged as _extras_bytes - # below, a local one by stat and a remote one from its own listing (or the - # flat reserve). Charging the target repository's sidecar on top of either - # is the double count that 409s a load which fits. The earlier local-only - # form predates the remote pricing and would now over-charge. - _extras_own_drafter = bool(_extra_args_own_spec and _extras_own_draft_path) + # An extras draft path wins whether or not the extras also own --spec-type: + # the loader ranks _cli_draft_for_budget ahead of _studio_draft_for_budget, + # and the launch appends the caller's flags after Studio's, so last-wins + # leaves exactly one --model-draft resident. It is charged as _extras_bytes + # below, local by stat and remote from its own listing, so charging the + # repository's sidecar on top is the double count that 409s a load that fits. + _extras_own_drafter = bool(_extras_own_draft_path) + # -ngld 0 / --spec-draft-device cpu applies to whichever separate drafter + # launches, including one Studio discovered itself, so none of them belongs + # in a VRAM budget. An embedded head is unaffected by draft-only flags and + # is inside the main weights either way, so nothing here suppresses it. + _draft_pinned_to_cpu = _extra_args_draft_offloaded_to_cpu( + llama_extra_args, env = os.environ + ) _forced_dspark = bool( (_spec_mode == "dspark" or _extra_args_requests_dspark(llama_extra_args, env = {})) and not _extras_own_drafter + and not _draft_pinned_to_cpu ) # Auto loads the sidecar whenever the model has one, so size it there too # or the guard admits a load 11 GB larger than it estimated. - _auto_dspark = _spec_mode == "auto" and not _extras_own_drafter + _auto_dspark = ( + _spec_mode == "auto" and not _extras_own_drafter and not _draft_pinned_to_cpu + ) _dspark_capable = True if _forced_dspark or _auto_dspark: # Gate on the same answer the loader uses: _download_dspark skips the @@ -6059,8 +6073,11 @@ def _estimate_gguf_required_gb( or (_spec_mode == "dflash" and not _extra_args_own_spec) ) and not _extras_own_drafter + and not _draft_pinned_to_cpu + ) + _auto_dflash = ( + _spec_mode == "auto" and not _extra_args_own_spec and not _draft_pinned_to_cpu ) - _auto_dflash = _spec_mode == "auto" and not _extra_args_own_spec _dflash_capable = True if _forced_dflash or _auto_dflash: try: @@ -6080,7 +6097,8 @@ def _estimate_gguf_required_gb( # that fits. Auto is different: it falls through to the MTP branch, and keeps # its charge. _charge_no_drafter = ( - _extras_own_drafter + _draft_pinned_to_cpu + or _extras_own_drafter or (_forced_dspark and not _dspark_capable) or (_forced_dflash and not _dflash_capable) ) diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index d16a11089a..3206aa977b 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -1341,8 +1341,10 @@ def test_extra_args_drafter_is_charged_once_when_it_is_the_local_sidecar(self): """--model-draft usually names the very sidecar discovery already found, and charging it on both paths billed a 1.5 GiB drafter as 3 GiB, so the guard refused an inference load that fits. Identity is the resolved path, - so a symlink or another spelling of the same file dedupes too, while a - genuinely separate drafter outside the model directory is still charged. + so a symlink or another spelling of the same file dedupes too. A drafter + somewhere else is charged instead of the discovered sidecar, not on top of + it: the loader ranks the extras path ahead of Studio's and the launch + appends the caller's flags last, so only one --model-draft is resident. """ import os import tempfile @@ -1390,7 +1392,8 @@ def test_extra_args_drafter_is_charged_once_when_it_is_the_local_sidecar(self): self.assertAlmostEqual(plain, 5000 / (1024**3), places = 9) self.assertAlmostEqual(same, 5000 / (1024**3), places = 9) # not 8000 self.assertAlmostEqual(through_link, 5000 / (1024**3), places = 9) - self.assertAlmostEqual(separate, 9000 / (1024**3), places = 9) # 2000+3000+4000 + # 2000 target + 4000 override; the 3000 sidecar loses to it and is not charged. + self.assertAlmostEqual(separate, 6000 / (1024**3), places = 9) def test_extras_owning_spec_type_charge_only_their_own_drafter(self): """--spec-type in the extras ends _build_speculative_flags before discovery's @@ -1936,6 +1939,77 @@ def test_a_cpu_offloaded_extras_drafter_is_not_charged_vram(self): ) self.assertAlmostEqual(on_gpu - on_cpu, 8.0, places = 6) + def test_a_bare_model_draft_overrides_the_repository_sidecar(self): + """No --spec-type needed for the override to win: the loader ranks the + extras draft path ahead of Studio's and the launch appends the caller's + flags last, so exactly one --model-draft is resident. Charging the repo's + sidecar as well 409s a load that fits.""" + import tempfile + + import utils.models.model_config as mc + + with tempfile.TemporaryDirectory() as d: + target = Path(d) / "model.gguf" + sidecar = Path(d) / "dspark-model-Q8_0.gguf" + custom = Path(d) / "elsewhere.gguf" + target.write_bytes(b"t" * 2000) + sidecar.write_bytes(b"s" * 3000) + custom.write_bytes(b"c" * 4000) + cfg = SimpleNamespace( + gguf_file = str(target), + gguf_mmproj_file = None, + gguf_mtp_file = None, + gguf_dspark_file = str(sidecar), + gguf_dflash_file = None, + gguf_hf_repo = None, + gguf_variant = None, + ) + with ( + patch.object(self.route, "_estimate_gguf_kv_gb", return_value = 0.0), + self._dspark_capable(), + ): + charged = self.route._estimate_gguf_required_gb( + cfg, + speculative_type = "auto", + llama_extra_args = ["--model-draft", str(custom)], + ) + # 2000 target + 4000 override, not the 3000 sidecar it displaces. + self.assertAlmostEqual(charged, 6000 / (1024**3), places = 9) + + def test_a_cpu_pinned_discovered_sidecar_is_not_charged_vram(self): + """-ngld 0 applies to whichever separate drafter launches, including one + Studio resolved itself with no draft path in the extras at all. It is then + host-resident, so charging it against the training job's VRAM 409s a load + that takes none.""" + import tempfile + + with tempfile.TemporaryDirectory() as d: + target = Path(d) / "model.gguf" + sidecar = Path(d) / "dspark-model-Q8_0.gguf" + target.write_bytes(b"t" * 2000) + sidecar.write_bytes(b"s" * 3000) + cfg = SimpleNamespace( + gguf_file = str(target), + gguf_mmproj_file = None, + gguf_mtp_file = None, + gguf_dspark_file = str(sidecar), + gguf_dflash_file = None, + gguf_hf_repo = None, + gguf_variant = None, + ) + with ( + patch.object(self.route, "_estimate_gguf_kv_gb", return_value = 0.0), + self._dspark_capable(), + ): + on_gpu = self.route._estimate_gguf_required_gb(cfg, speculative_type = "auto") + on_cpu = self.route._estimate_gguf_required_gb( + cfg, + speculative_type = "auto", + llama_extra_args = ["--spec-draft-ngl", "0"], + ) + self.assertAlmostEqual(on_gpu, 5000 / (1024**3), places = 9) + self.assertAlmostEqual(on_cpu, 2000 / (1024**3), places = 9) + def test_a_partial_dflash_shard_set_is_not_charged(self): """The fetch refuses a family whose encoded shard count is short, so a listing caught mid-publication must not be billed for its listed half: From 4a926886251a25f51b5d96bf6db62c5f99b51283 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Tue, 11 Aug 2026 21:20:34 +0000 Subject: [PATCH 17/25] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/routes/inference.py | 8 ++------ 1 file changed, 2 insertions(+), 6 deletions(-) diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 19c1adc6d5..9fa2368575 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -6031,9 +6031,7 @@ def _estimate_gguf_required_gb( # launches, including one Studio discovered itself, so none of them belongs # in a VRAM budget. An embedded head is unaffected by draft-only flags and # is inside the main weights either way, so nothing here suppresses it. - _draft_pinned_to_cpu = _extra_args_draft_offloaded_to_cpu( - llama_extra_args, env = os.environ - ) + _draft_pinned_to_cpu = _extra_args_draft_offloaded_to_cpu(llama_extra_args, env = os.environ) _forced_dspark = bool( (_spec_mode == "dspark" or _extra_args_requests_dspark(llama_extra_args, env = {})) and not _extras_own_drafter @@ -6041,9 +6039,7 @@ def _estimate_gguf_required_gb( ) # Auto loads the sidecar whenever the model has one, so size it there too # or the guard admits a load 11 GB larger than it estimated. - _auto_dspark = ( - _spec_mode == "auto" and not _extras_own_drafter and not _draft_pinned_to_cpu - ) + _auto_dspark = _spec_mode == "auto" and not _extras_own_drafter and not _draft_pinned_to_cpu _dspark_capable = True if _forced_dspark or _auto_dspark: # Gate on the same answer the loader uses: _download_dspark skips the From 02b2314abeba68c1df8aea2ecaf9c828f1f2d20d Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Tue, 11 Aug 2026 21:59:54 +0000 Subject: [PATCH 18/25] Charge nothing for a drafter repo whose every split is incomplete, per directory --- studio/backend/routes/inference.py | 9 ++++++++- .../backend/tests/test_chat_load_during_training.py | 13 +++++++++++++ studio/backend/tests/test_mtp_drafter_companion.py | 13 +++++++++++++ studio/backend/utils/models/drafters/common.py | 11 ++++++++++- 4 files changed, 44 insertions(+), 2 deletions(-) diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 9fa2368575..d13844b417 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5754,7 +5754,14 @@ def _remote_drafter_repo_bytes(spec: str, *, hf_token: Optional[str]) -> int: # the basename restores every quant and charges the repo's F16. matched = {n: s for n, s in sizes.items() if hint in n.lower()} sizes = matched or sizes - return dflash_budget_bytes(sizes, _gguf_extra_shards) or _REMOTE_DRAFTER_RESERVE_BYTES + # An empty listing is "we learned nothing" and falls through to the cache + # and then the reserve. A listing that DID name GGUFs and still bounds at + # zero is different: every family was rejected as an incomplete split, so + # the fetch can load none of them and no draft weights become resident. + # Charging the reserve there 409s a load for VRAM nothing will take. + bounded = dflash_budget_bytes(sizes, _gguf_extra_shards) + if bounded or sizes: + return bounded except Exception as e: logger.warning(f"Could not size remote drafter repo {spec}: {e}") # Unreadable listings are the case where the repo is already in the local HF diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index 3206aa977b..a31335e295 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -2010,6 +2010,19 @@ def test_a_cpu_pinned_discovered_sidecar_is_not_charged_vram(self): self.assertAlmostEqual(on_gpu, 5000 / (1024**3), places = 9) self.assertAlmostEqual(on_cpu, 2000 / (1024**3), places = 9) + def test_a_drafter_repo_with_only_a_partial_set_is_charged_nothing(self): + """The bound is deliberately zero when every family is an incomplete split: + the fetch refuses all of them, so no draft weights become resident. Turning + that into the unreadable-listing reserve 409s a load for VRAM nothing takes.""" + partial = SimpleNamespace( + siblings = [ + SimpleNamespace(rfilename = "drafter-00001-of-00002.gguf", size = 4 * 1024**3) + ] + ) + with patch("huggingface_hub.model_info", return_value = partial): + charged = self.route._remote_drafter_repo_bytes("org/drafter", hf_token = None) + self.assertEqual(charged, 0) + def test_a_partial_dflash_shard_set_is_not_charged(self): """The fetch refuses a family whose encoded shard count is short, so a listing caught mid-publication must not be billed for its listed half: diff --git a/studio/backend/tests/test_mtp_drafter_companion.py b/studio/backend/tests/test_mtp_drafter_companion.py index d6317447e7..a3f05f8fe7 100644 --- a/studio/backend/tests/test_mtp_drafter_companion.py +++ b/studio/backend/tests/test_mtp_drafter_companion.py @@ -3462,3 +3462,16 @@ def test_download_dflash_reaches_the_complete_family_behind_an_incomplete_one( near_path = str(weight), ) assert picked == "dflash-kquant.gguf" + + +def test_split_completeness_is_scoped_to_the_files_own_directory(): + """A repo laid out by quant can hold half of one broken set beside half of + another. Matching basenames alone calls both of them one whole set, and the + fetch then cannot load either.""" + from utils.models.drafters import split_listing_is_complete + + names = ["Q4/model-00001-of-00002.gguf", "Q8/model-00002-of-00002.gguf"] + assert not split_listing_is_complete(names, names[0]) + assert not split_listing_is_complete(names, names[1]) + whole = ["Q4/model-00001-of-00002.gguf", "Q4/model-00002-of-00002.gguf"] + assert split_listing_is_complete(whole, whole[0]) diff --git a/studio/backend/utils/models/drafters/common.py b/studio/backend/utils/models/drafters/common.py index 1502448aaa..5258fdf016 100644 --- a/studio/backend/utils/models/drafters/common.py +++ b/studio/backend/utils/models/drafters/common.py @@ -153,13 +153,22 @@ def split_listing_is_complete(names: Iterable[str], name: str) -> bool: The listing counterpart of _drafter_split_is_complete, which needs files on disk. A repo mid-upload lists part of a set and the fetch refuses that, so the plan and the budget must agree. True for a single-file name, which encodes no set. + + Counted within the file's own directory. A repo laid out by quant can hold + Q4/model-00001-of-00002.gguf beside Q8/model-00002-of-00002.gguf, and matching + on basenames alone would call both halves of two broken sets one whole one. """ match = _LISTED_SHARD_RE.match(Path(name).name) if not match: return True stem, total = match.group(1), match.group(3) + parent = Path(name).parent sibling = re.compile( r"^" + re.escape(stem) + r"-\d{5}-of-" + re.escape(total) + r"\.gguf$", re.IGNORECASE, ) - return sum(1 for other in names if sibling.match(Path(other).name)) == int(total) + return sum( + 1 + for other in names + if Path(other).parent == parent and sibling.match(Path(other).name) + ) == int(total) From a2509138441e6974a7a573e151234d1886c5f541 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Tue, 11 Aug 2026 22:01:00 +0000 Subject: [PATCH 19/25] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/tests/test_chat_load_during_training.py | 4 +--- studio/backend/utils/models/drafters/common.py | 4 +--- 2 files changed, 2 insertions(+), 6 deletions(-) diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index a31335e295..3e245f3300 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -2015,9 +2015,7 @@ def test_a_drafter_repo_with_only_a_partial_set_is_charged_nothing(self): the fetch refuses all of them, so no draft weights become resident. Turning that into the unreadable-listing reserve 409s a load for VRAM nothing takes.""" partial = SimpleNamespace( - siblings = [ - SimpleNamespace(rfilename = "drafter-00001-of-00002.gguf", size = 4 * 1024**3) - ] + siblings = [SimpleNamespace(rfilename = "drafter-00001-of-00002.gguf", size = 4 * 1024**3)] ) with patch("huggingface_hub.model_info", return_value = partial): charged = self.route._remote_drafter_repo_bytes("org/drafter", hf_token = None) diff --git a/studio/backend/utils/models/drafters/common.py b/studio/backend/utils/models/drafters/common.py index 5258fdf016..b7be6bce2c 100644 --- a/studio/backend/utils/models/drafters/common.py +++ b/studio/backend/utils/models/drafters/common.py @@ -168,7 +168,5 @@ def split_listing_is_complete(names: Iterable[str], name: str) -> bool: re.IGNORECASE, ) return sum( - 1 - for other in names - if Path(other).parent == parent and sibling.match(Path(other).name) + 1 for other in names if Path(other).parent == parent and sibling.match(Path(other).name) ) == int(total) From 65e08f67a8e32c570a891b1f6a90d1b5bca1bb89 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Tue, 11 Aug 2026 22:29:31 +0000 Subject: [PATCH 20/25] Studio: narrow the drafter VRAM charges and keep the MLA drop reason - --model-draft naming a path that is not on disk is no longer priced as a Hugging Face repo, so a typo cannot 409 a chat load over a 12 GiB reserve. - A drafter listing that carries no sizes falls through to the cache and then the reserve; zero is kept for the case where every family is an incomplete split and the fetch can load none of them. - Auto DFlash is not charged when the extras already name their own drafter. - An MLA embedded-MTP model keeps mla_mtp_disabled as its fallback reason, so the notice does not invite the user to force a path that is slower than the ngram-mod they are getting. --- studio/backend/core/inference/llama_cpp.py | 12 ++- studio/backend/routes/inference.py | 40 ++++++++- .../tests/test_chat_load_during_training.py | 83 +++++++++++++++++++ .../backend/tests/test_llama_cpp_placement.py | 24 ++++++ 4 files changed, 154 insertions(+), 5 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 791c750f90..022999e5d6 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -14150,7 +14150,17 @@ def _fallback_drafter_not_found() -> None: # UD-Q4_K_XL + mmproj-kquant + dflash-kquant, b10342, B200, n_max=2, greedy, # ~545 image tokens gave 92.1 -> 114.2 tok/s at 0.646 acceptance with output # byte-identical to the drafter-free run. - if drafter_no_vram: + # Which drafter Auto would have emitted had VRAM allowed. An MLA model with + # no sidecar is already dropped by policy below, so reporting the VRAM reason + # for it would tell the user to force MTP at a smaller context: a path that is + # slower than the ngram-mod they are getting. Let it fall through to the MLA + # branch, which drops the same drafter for the reason that actually applies. + _no_vram_drops_a_real_drafter = ( + bool(dspark_draft_path and caps.get("supports_dspark")) + or bool(dflash_draft_path and caps.get("supports_dflash")) + or not _auto_mla_embedded_mtp + ) + if drafter_no_vram and _no_vram_drops_a_real_drafter: # The fit found room for the target but not for the drafter's reserve, # and reserved nothing for it, so emitting one now would OOM the load. # Same downgrade as the MLA branch below: ngram-mod costs no VRAM, and diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index d13844b417..42f127a446 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5686,6 +5686,22 @@ def _remote_gguf_companion_bytes( # established direction here is to over-estimate. Only reached when the listing # cannot be read at all, where llama-server may still open the repo from its # local HF cache and make every one of those bytes resident. +# The extras flags that name a Hugging Face repository rather than a path on disk. +# _extra_args_mtp_draft_path returns the value without saying which flag carried it, +# and the two need different treatment: a repo id is sized from its listing, while a +# --model-draft that is not a file is simply a drafter that will not load. +_HF_DRAFT_FLAGS = ("--spec-draft-hf", "-hfd", "-hfrd", "--hf-repo-draft") + + +def _extra_args_name_a_remote_draft(extra_args: Optional[list[str]]) -> bool: + for raw in extra_args or []: + token = str(raw) + head = token.split("=", 1)[0] + if head in _HF_DRAFT_FLAGS: + return True + return False + + _REMOTE_DRAFTER_RESERVE_BYTES = 12 * 1024**3 @@ -5734,7 +5750,7 @@ def _remote_drafter_repo_bytes(spec: str, *, hf_token: Optional[str]) -> int: try: from core.inference.llama_cpp import _gguf_extra_shards from huggingface_hub import model_info - from utils.models.drafters import dflash_budget_bytes + from utils.models.drafters import dflash_budget_bytes, split_listing_is_complete info = model_info(repo, token = hf_token, files_metadata = True) sizes: dict[str, int] = {} @@ -5760,8 +5776,15 @@ def _remote_drafter_repo_bytes(spec: str, *, hf_token: Optional[str]) -> int: # the fetch can load none of them and no draft weights become resident. # Charging the reserve there 409s a load for VRAM nothing will take. bounded = dflash_budget_bytes(sizes, _gguf_extra_shards) - if bounded or sizes: + if bounded: return bounded + # Zero from a non-empty listing means one of two things. Every family + # incomplete: the fetch can load none of them, so zero is the answer. + # Complete families whose sizes the listing did not carry: something WILL + # load and we simply do not know how big, which is the cache-then-reserve + # case, not a free load. + if sizes and not any(split_listing_is_complete(sizes, name) for name in sizes): + return 0 except Exception as e: logger.warning(f"Could not size remote drafter repo {spec}: {e}") # Unreadable listings are the case where the repo is already in the local HF @@ -6078,8 +6101,14 @@ def _estimate_gguf_required_gb( and not _extras_own_drafter and not _draft_pinned_to_cpu ) + # Two independent ways Auto never reaches DFlash: extras owning --spec-type + # return before any mode branch, and extras naming their own drafter are the + # drafter the loader uses instead of a discovered sidecar. _auto_dflash = ( - _spec_mode == "auto" and not _extra_args_own_spec and not _draft_pinned_to_cpu + _spec_mode == "auto" + and not _extra_args_own_spec + and not _extras_own_drafter + and not _draft_pinned_to_cpu ) _dflash_capable = True if _forced_dflash or _auto_dflash: @@ -6186,8 +6215,11 @@ def _same_file_key(p: str) -> str: elif _extras_draft and Path(_extras_draft).is_file(): if _same_file_key(str(_extras_draft)) not in _sized_keys: _extras_bytes = LlamaCppBackend._get_gguf_size_bytes(str(_extras_draft)) - elif _extras_draft: + elif _extras_draft and _extra_args_name_a_remote_draft(llama_extra_args): _extras_bytes = _remote_drafter_repo_bytes(str(_extras_draft), hf_token = hf_token) + # else: a local --model-draft that is not on disk. llama-server cannot load + # a drafter from it, so it becomes no VRAM; charging a repository reserve + # for a misspelled path would 409 a load whose drafter simply never starts. if total_bytes > 0: return (total_bytes + _extras_bytes) / (1024**3) + _estimate_gguf_kv_gb( diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index 3e245f3300..23bc3d9359 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -1871,6 +1871,89 @@ def test_an_unlistable_remote_drafter_is_measured_from_the_local_cache(self): # The whole 30 GB set that is actually resident, not the flat reserve. self.assertEqual(charged, 30 * 1024**3) + def test_a_local_model_draft_that_is_not_on_disk_is_not_priced_as_a_repo(self): + """--model-draft takes a path, --spec-draft-hf takes a repo id. A path that + does not exist is a drafter llama-server will not load, so it costs nothing. + Charging it the unreadable-repo reserve 409s the chat load over 12 GiB that + a typo, not a download, put in the extras.""" + import utils.models.model_config as mc + + cfg = SimpleNamespace( + gguf_file = None, + gguf_mmproj_file = None, + gguf_mtp_file = None, + gguf_dspark_file = None, + gguf_dflash_file = None, + gguf_hf_repo = "org/repo", + gguf_variant = "Q4_K_M", + ) + variant = SimpleNamespace(quant = "Q4_K_M", size_bytes = 4 * 1024**3) + with ( + patch.object(mc, "list_gguf_variants", lambda repo, hf_token = None: ([variant], False)), + patch.object(self.route, "_remote_gguf_companion_bytes", return_value = 0), + patch.object(self.route, "_estimate_gguf_kv_gb", return_value = 0.0), + patch("huggingface_hub.model_info", side_effect = AssertionError("priced as a repo")), + ): + charged = self.route._estimate_gguf_required_gb( + cfg, + speculative_type = "auto", + llama_extra_args = ["--model-draft", "/nope/does-not-exist.gguf"], + ) + self.assertAlmostEqual(charged, 4.0, places = 6) + + def test_a_listing_that_carries_no_sizes_still_pays_the_reserve(self): + """A complete family whose listing omits sizes is not a free drafter, it is + an unmeasured one: something loads and the guard does not know how big. The + zero answer belongs to the case where every family is an incomplete split + and the fetch can load none of them.""" + sizeless = SimpleNamespace( + siblings = [SimpleNamespace(rfilename = "drafter-Q4_K_M.gguf", size = None)] + ) + with ( + patch("huggingface_hub.model_info", return_value = sizeless), + patch("huggingface_hub.scan_cache_dir", return_value = SimpleNamespace(repos = [])), + ): + charged = self.route._remote_drafter_repo_bytes("org/drafter", hf_token = None) + self.assertEqual(charged, self.route._REMOTE_DRAFTER_RESERVE_BYTES) + + def test_extras_naming_their_own_drafter_do_not_also_pay_for_a_dflash_sidecar(self): + """A drafter in the extras is the drafter the loader launches: the Auto + promotion that would have discovered the repo's DFlash sidecar never runs. + Billing both charges the training job for weights only one of them loads.""" + import utils.models.model_config as mc + + cfg = SimpleNamespace( + gguf_file = None, + gguf_mmproj_file = None, + gguf_mtp_file = None, + gguf_dspark_file = None, + gguf_dflash_file = None, + gguf_hf_repo = "org/repo", + gguf_variant = "Q4_K_M", + ) + variant = SimpleNamespace( + filename = "model-Q4_K_M.gguf", quant = "Q4_K_M", size_bytes = 10 * 1024**3 + ) + siblings = [SimpleNamespace(rfilename = "dflash-kquant.gguf", size = 2 * 1024**3)] + with ( + patch.object(mc, "list_gguf_variants", lambda repo, hf_token = None: ([variant], False)), + patch( + "huggingface_hub.model_info", + return_value = SimpleNamespace(siblings = siblings), + ), + patch.object(self.route, "_remote_drafter_repo_bytes", return_value = 3 * 1024**3), + patch.object(self.route, "_estimate_gguf_kv_gb", return_value = 0.0), + self._dflash_capable(), + ): + charged = self.route._estimate_gguf_required_gb( + cfg, + speculative_type = "auto", + llama_extra_args = ["--spec-draft-hf", "org/drafter"], + ) + # The extras drafter, not the extras drafter plus the sidecar Auto no + # longer reaches. + self.assertAlmostEqual(charged, 13.0, places = 6) + def test_a_remote_drafter_that_is_neither_listable_nor_cached_pays_the_reserve(self): """Nothing to measure and nothing to download over the Hub that just refused the listing, so the reserve is a cushion rather than a bound.""" diff --git a/studio/backend/tests/test_llama_cpp_placement.py b/studio/backend/tests/test_llama_cpp_placement.py index 3e1c20b7cc..9c8a61143e 100644 --- a/studio/backend/tests/test_llama_cpp_placement.py +++ b/studio/backend/tests/test_llama_cpp_placement.py @@ -557,6 +557,30 @@ def test_a_busy_second_gpu_does_not_condemn_a_drafter_the_first_one_holds(tmp_pa assert backend.spec_fallback_reason is None +def test_an_mla_model_keeps_the_reason_that_actually_dropped_its_drafter(tmp_path): + """An MLA embedded-MTP model has no drafter to save: Auto drops it by policy, + because llama.cpp's MLA/DSA MTP path is slower than no speculation at all. If + the VRAM branch claims it first, the notice tells the user to force MTP at a + smaller context, i.e. to buy a known regression with their context length.""" + backend, gguf, _sidecar = _tight_vram_backend(tmp_path, drafter_gb = 12.0) + # Embedded head, MLA geometry, no sidecar: exactly the GLM-5.2 shape. + backend._nextn_predict_layers = 1 + backend._kv_lora_rank = 512 + backend.probe_server_capabilities = lambda _binary = None: { + "supports_mtp": True, + "mtp_token": "mtp", + "supports_ngram_mod": True, + "spec_draft_n_max_flag": "--spec-draft-n-max", + } + + result = _launch(backend, gguf, speculative_type = "auto", n_ctx = 8192) + + cmd = result["cmd"] + assert "draft-mtp" not in cmd + assert cmd[cmd.index("--spec-type") + 1] == "ngram-mod" + assert backend.spec_fallback_reason == "mla_mtp_disabled" + + def test_tensor_parallel_keeps_its_own_sizing(tmp_path): """_plan_tensor_parallel reserves a per-device tensor buffer on geometry this layer-split probe does not model, so under tensor mode the probe stands down From eb7549c345020e6b1b61e13860da1547d915ff69 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Wed, 12 Aug 2026 12:23:02 +0000 Subject: [PATCH 21/25] Studio: classify the winning draft flag and check shard indices - Draft flags are last-wins, so a repo id followed by --model-draft leaves a path as the drafter. _extra_args_mtp_draft_source now returns the value and whether the flag that won carried a repo id, so a path is never priced as a Hugging Face repository. - split_listing_is_complete tracks distinct shard indices inside 1..total rather than counting matches, so 00001-of-00002 beside a stray 00003-of-00002 is no longer read as a whole set. --- studio/backend/core/inference/llama_cpp.py | 56 ++++++++++++------- studio/backend/routes/inference.py | 26 +++------ .../tests/test_chat_load_during_training.py | 35 ++++++++++++ .../tests/test_mtp_drafter_companion.py | 16 ++++++ .../backend/utils/models/drafters/common.py | 22 ++++++-- 5 files changed, 112 insertions(+), 43 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 022999e5d6..0547af469c 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -2725,37 +2725,53 @@ def _extra_args_mtp_draft_path( --spec-draft-hf/-hfd/...), else the LLAMA_ARG_SPEC_DRAFT_MODEL/_HF_REPO env, else None. An HF repo isn't a local file, so the budget can't size it (falls back to the flat reserve), but recognizing it avoids sizing the wrong one.""" - flags = { - "--model-draft", - "--spec-draft-model", - "-md", - "--spec-draft-hf", - "-hfd", - "-hfrd", - "--hf-repo-draft", - } + return _extra_args_mtp_draft_source(extra_args, env)[0] + + +_HF_DRAFT_FLAGS = frozenset({"--spec-draft-hf", "-hfd", "-hfrd", "--hf-repo-draft"}) +_LOCAL_DRAFT_FLAGS = frozenset({"--model-draft", "--spec-draft-model", "-md"}) + + +def _extra_args_mtp_draft_source( + extra_args: Optional[Iterable[str]], env: Optional[Mapping[str, str]] = None +) -> tuple[Optional[str], bool]: + """The drafter the launch resolves to, and whether it is a repo id. + + The two travel together on purpose. Draft flags are last-wins, so a remote flag + followed by a local one leaves a local path as the winner; a caller that asks + only "does any remote flag appear" would then price a filesystem path as a + Hugging Face repository and charge a reserve for something llama-server will + not load. Only the winning flag decides which of the two the value is. + """ args = [str(a) for a in extra_args] if extra_args else [] found: Optional[str] = None + found_remote = False for i, raw in enumerate(args): flag = _flag_name(raw) _, eq, inline = raw.partition("=") - if flag not in flags: + if flag not in _HF_DRAFT_FLAGS and flag not in _LOCAL_DRAFT_FLAGS: continue value = inline if eq else (args[i + 1] if i + 1 < len(args) else "") if value and not value.startswith("-"): found = value + found_remote = flag in _HF_DRAFT_FLAGS if found is not None: - return found + return found, found_remote e = os.environ if env is None else env - return ( - e.get("LLAMA_ARG_SPEC_DRAFT_MODEL") - or e.get("LLAMA_ARG_SPEC_DRAFT_HF_REPO") - # Pre-b8955 spellings, live between the launchable floor (2025-08-30) and the - # rename (2026-04-28). - or e.get("LLAMA_ARG_MODEL_DRAFT") - or e.get("LLAMA_ARG_HFD_REPO") - or None - ) + # Same precedence the value lookup has always used; the env spelling is what + # says local or remote here, there being no flag to read. + # Pre-b8955 spellings, live between the launchable floor (2025-08-30) and the + # rename (2026-04-28). + for var, remote in ( + ("LLAMA_ARG_SPEC_DRAFT_MODEL", False), + ("LLAMA_ARG_SPEC_DRAFT_HF_REPO", True), + ("LLAMA_ARG_MODEL_DRAFT", False), + ("LLAMA_ARG_HFD_REPO", True), + ): + value = e.get(var) + if value: + return value, remote + return None, False def _extra_args_draft_cache_types( diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 42f127a446..e2c8440acf 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5686,22 +5686,6 @@ def _remote_gguf_companion_bytes( # established direction here is to over-estimate. Only reached when the listing # cannot be read at all, where llama-server may still open the repo from its # local HF cache and make every one of those bytes resident. -# The extras flags that name a Hugging Face repository rather than a path on disk. -# _extra_args_mtp_draft_path returns the value without saying which flag carried it, -# and the two need different treatment: a repo id is sized from its listing, while a -# --model-draft that is not a file is simply a drafter that will not load. -_HF_DRAFT_FLAGS = ("--spec-draft-hf", "-hfd", "-hfrd", "--hf-repo-draft") - - -def _extra_args_name_a_remote_draft(extra_args: Optional[list[str]]) -> bool: - for raw in extra_args or []: - token = str(raw) - head = token.split("=", 1)[0] - if head in _HF_DRAFT_FLAGS: - return True - return False - - _REMOTE_DRAFTER_RESERVE_BYTES = 12 * 1024**3 @@ -6028,6 +6012,7 @@ def _estimate_gguf_required_gb( _canonicalize_spec_mode, _extra_args_draft_offloaded_to_cpu, _extra_args_mtp_draft_path, + _extra_args_mtp_draft_source, _extra_args_requests_dflash, _extra_args_requests_dspark, _extra_args_set_spec_type, @@ -6205,7 +6190,12 @@ def _same_file_key(p: str) -> str: # sidecars, so a target that ships none left a multi-GB drafter billed # nowhere and the guard admitted a load that evicts the training job. _extras_bytes = 0 - _extras_draft = _extra_args_mtp_draft_path(llama_extra_args, env = {}) + # The value AND which flag carried it. Draft flags are last-wins, so a repo + # id followed by a path leaves a path as the drafter, and pricing that path + # as a repository would charge a reserve for something that never loads. + _extras_draft, _extras_draft_is_remote = _extra_args_mtp_draft_source( + llama_extra_args, env = {} + ) # -ngld 0 / --spec-draft-device cpu keeps the drafter in host memory, so it # competes for RAM, not for the training job's VRAM. Charging it here is not # the safe over-estimate the rest of this guard makes; it is simply the wrong @@ -6215,7 +6205,7 @@ def _same_file_key(p: str) -> str: elif _extras_draft and Path(_extras_draft).is_file(): if _same_file_key(str(_extras_draft)) not in _sized_keys: _extras_bytes = LlamaCppBackend._get_gguf_size_bytes(str(_extras_draft)) - elif _extras_draft and _extra_args_name_a_remote_draft(llama_extra_args): + elif _extras_draft and _extras_draft_is_remote: _extras_bytes = _remote_drafter_repo_bytes(str(_extras_draft), hf_token = hf_token) # else: a local --model-draft that is not on disk. llama-server cannot load # a drafter from it, so it becomes no VRAM; charging a repository reserve diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index 23bc3d9359..c06891a8cd 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -1901,6 +1901,41 @@ def test_a_local_model_draft_that_is_not_on_disk_is_not_priced_as_a_repo(self): ) self.assertAlmostEqual(charged, 4.0, places = 6) + def test_only_the_winning_draft_flag_decides_repo_or_path(self): + """Draft flags are last-wins in llama-server, so a repo id followed by a + --model-draft leaves the path as the drafter. Asking "does any remote flag + appear" prices that path as a repository and charges the 12 GiB reserve for + a drafter the launch cannot open.""" + import utils.models.model_config as mc + + cfg = SimpleNamespace( + gguf_file = None, + gguf_mmproj_file = None, + gguf_mtp_file = None, + gguf_dspark_file = None, + gguf_dflash_file = None, + gguf_hf_repo = "org/repo", + gguf_variant = "Q4_K_M", + ) + variant = SimpleNamespace(quant = "Q4_K_M", size_bytes = 4 * 1024**3) + with ( + patch.object(mc, "list_gguf_variants", lambda repo, hf_token = None: ([variant], False)), + patch.object(self.route, "_remote_gguf_companion_bytes", return_value = 0), + patch.object(self.route, "_estimate_gguf_kv_gb", return_value = 0.0), + patch("huggingface_hub.model_info", side_effect = AssertionError("priced as a repo")), + ): + charged = self.route._estimate_gguf_required_gb( + cfg, + speculative_type = "auto", + llama_extra_args = [ + "--spec-draft-hf", + "org/drafter", + "--model-draft", + "/nope/does-not-exist.gguf", + ], + ) + self.assertAlmostEqual(charged, 4.0, places = 6) + def test_a_listing_that_carries_no_sizes_still_pays_the_reserve(self): """A complete family whose listing omits sizes is not a free drafter, it is an unmeasured one: something loads and the guard does not know how big. The diff --git a/studio/backend/tests/test_mtp_drafter_companion.py b/studio/backend/tests/test_mtp_drafter_companion.py index a3f05f8fe7..92f9be7619 100644 --- a/studio/backend/tests/test_mtp_drafter_companion.py +++ b/studio/backend/tests/test_mtp_drafter_companion.py @@ -3464,6 +3464,22 @@ def test_download_dflash_reaches_the_complete_family_behind_an_incomplete_one( assert picked == "dflash-kquant.gguf" +def test_split_completeness_reads_shard_indices_not_a_shard_count(): + """A listing caught mid-publication can hold 00001-of-00002 beside a stray + 00003-of-00002. Two files, so a count calls the set whole, while shard 2 is + still missing and llama-server cannot open it: the picker then ranks a family + it cannot load and the guard bills the training job for it.""" + from utils.models.drafters import split_listing_is_complete + + names = ["model-00001-of-00002.gguf", "model-00003-of-00002.gguf"] + assert not split_listing_is_complete(names, names[0]) + # A duplicate listing of one shard is the same trap without the odd index. + dupes = ["dir/model-00001-of-00002.gguf", "dir/model-00001-of-00002.gguf"] + assert not split_listing_is_complete(dupes, dupes[0]) + whole = ["model-00001-of-00002.gguf", "model-00002-of-00002.gguf"] + assert split_listing_is_complete(whole, whole[0]) + + def test_split_completeness_is_scoped_to_the_files_own_directory(): """A repo laid out by quant can hold half of one broken set beside half of another. Matching basenames alone calls both of them one whole set, and the diff --git a/studio/backend/utils/models/drafters/common.py b/studio/backend/utils/models/drafters/common.py index b7be6bce2c..b79795e22d 100644 --- a/studio/backend/utils/models/drafters/common.py +++ b/studio/backend/utils/models/drafters/common.py @@ -161,12 +161,24 @@ def split_listing_is_complete(names: Iterable[str], name: str) -> bool: match = _LISTED_SHARD_RE.match(Path(name).name) if not match: return True - stem, total = match.group(1), match.group(3) + stem, total = match.group(1), int(match.group(3)) parent = Path(name).parent sibling = re.compile( - r"^" + re.escape(stem) + r"-\d{5}-of-" + re.escape(total) + r"\.gguf$", + r"^" + re.escape(stem) + r"-(\d{5})-of-" + re.escape(match.group(3)) + r"\.gguf$", re.IGNORECASE, ) - return sum( - 1 for other in names if Path(other).parent == parent and sibling.match(Path(other).name) - ) == int(total) + # Distinct indices inside 1..total, not a count: a mid-publication listing can + # hold 00001-of-00002 beside a stray 00003-of-00002, and counting would call + # that pair whole while shard 2 is still missing and llama-server cannot open + # the set. _drafter_split_is_complete answers the on-disk version the same way. + seen = set() + for other in names: + if Path(other).parent != parent: + continue + found = sibling.match(Path(other).name) + if not found: + continue + index = int(found.group(1)) + if 1 <= index <= total: + seen.add(index) + return len(seen) == total From c0d1f537baa66ae294025d8c982526be3fa1f35d Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Wed, 12 Aug 2026 12:24:37 +0000 Subject: [PATCH 22/25] Studio: scan the Hugging Face cache Studio is pointed at _cached_repo_gguf_bytes scanned huggingface_hub's import-time default, so a user who moved the cache had their cached drafter missed and priced at the flat reserve instead. Pass the active hub cache, as the rest of Studio's cache operations do. Also covers the underscore flag spelling llama.cpp accepts. --- studio/backend/routes/inference.py | 13 +++- .../tests/test_chat_load_during_training.py | 62 +++++++++++++++++++ 2 files changed, 74 insertions(+), 1 deletion(-) diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index e2c8440acf..d0944e47f6 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5798,8 +5798,19 @@ def _cached_repo_gguf_bytes(repo: str, hint: str = "") -> int: from core.inference.llama_cpp import _gguf_extra_shards from utils.models.drafters import dflash_budget_bytes + # The cache Studio is pointed at right now, not the one huggingface_hub + # resolved at import. A user who moved the cache launches llama-server + # against the new one, so a bare scan misses the drafter that will load + # and falls back to a reserve that can undercount it. + try: + from utils.hf_cache_settings import active_hf_hub_cache + + _cache_dir: Optional[str] = active_hf_hub_cache() + except Exception: + _cache_dir = None + sizes: dict[str, int] = {} - for cached_repo in scan_cache_dir().repos: + for cached_repo in scan_cache_dir(cache_dir = _cache_dir).repos: if (cached_repo.repo_id or "").lower() != repo.lower(): continue # One revision, not every snapshot on disk. llama-server resolves the diff --git a/studio/backend/tests/test_chat_load_during_training.py b/studio/backend/tests/test_chat_load_during_training.py index c06891a8cd..68d36f14fb 100644 --- a/studio/backend/tests/test_chat_load_during_training.py +++ b/studio/backend/tests/test_chat_load_during_training.py @@ -1936,6 +1936,68 @@ def test_only_the_winning_draft_flag_decides_repo_or_path(self): ) self.assertAlmostEqual(charged, 4.0, places = 6) + def test_the_cached_drafter_scan_reads_the_cache_studio_is_pointed_at(self): + """A user who moved the Hugging Face cache launches llama-server against the + new one. Scanning huggingface_hub's import-time default finds nothing there, + and the reserve that replaces the measurement can undercount a large cached + drafter beside a running training job.""" + seen = {} + cached = SimpleNamespace( + repo_id = "org/drafter", + revisions = [ + SimpleNamespace( + refs = {"main"}, + last_modified = 2.0, + files = [ + SimpleNamespace(file_name = "drafter-Q4_K_M.gguf", size_on_disk = 7 * 1024**3) + ], + ) + ], + ) + + def fake_scan(cache_dir = None, **kwargs): + seen["cache_dir"] = cache_dir + return SimpleNamespace(repos = [cached]) + + with ( + patch("huggingface_hub.model_info", side_effect = OSError("no network")), + patch("huggingface_hub.scan_cache_dir", side_effect = fake_scan), + patch("utils.hf_cache_settings.active_hf_hub_cache", return_value = "/elsewhere/hub"), + ): + charged = self.route._remote_drafter_repo_bytes("org/drafter", hf_token = None) + self.assertEqual(seen["cache_dir"], "/elsewhere/hub") + self.assertEqual(charged, 7 * 1024**3) + + def test_underscore_spelled_draft_flags_classify_as_remote(self): + """llama.cpp accepts --spec_draft_hf as well as --spec-draft-hf. The value + parser normalises the spelling, so a classifier that compares raw tokens + calls the repo a local path, charges nothing, and lets the guard admit a + load whose drafter is multiple GB.""" + import utils.models.model_config as mc + + cfg = SimpleNamespace( + gguf_file = None, + gguf_mmproj_file = None, + gguf_mtp_file = None, + gguf_dspark_file = None, + gguf_dflash_file = None, + gguf_hf_repo = "org/repo", + gguf_variant = "Q4_K_M", + ) + variant = SimpleNamespace(quant = "Q4_K_M", size_bytes = 4 * 1024**3) + with ( + patch.object(mc, "list_gguf_variants", lambda repo, hf_token = None: ([variant], False)), + patch.object(self.route, "_remote_gguf_companion_bytes", return_value = 0), + patch.object(self.route, "_estimate_gguf_kv_gb", return_value = 0.0), + patch.object(self.route, "_remote_drafter_repo_bytes", return_value = 6 * 1024**3), + ): + charged = self.route._estimate_gguf_required_gb( + cfg, + speculative_type = "auto", + llama_extra_args = ["--spec_draft_hf", "org/drafter"], + ) + self.assertAlmostEqual(charged, 10.0, places = 6) + def test_a_listing_that_carries_no_sizes_still_pays_the_reserve(self): """A complete family whose listing omits sizes is not a free drafter, it is an unmeasured one: something loads and the guard does not know how big. The From 40bcaa1e6dff38f3aaee05f3a131af9857bc5413 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Wed, 12 Aug 2026 12:30:23 +0000 Subject: [PATCH 23/25] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/routes/inference.py | 1 - 1 file changed, 1 deletion(-) diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index d0944e47f6..60aef2f24e 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5804,7 +5804,6 @@ def _cached_repo_gguf_bytes(repo: str, hint: str = "") -> int: # and falls back to a reserve that can undercount it. try: from utils.hf_cache_settings import active_hf_hub_cache - _cache_dir: Optional[str] = active_hf_hub_cache() except Exception: _cache_dir = None From 0c1cd3753b2125625826ceeab746a4128457d2d9 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Wed, 12 Aug 2026 12:51:51 +0000 Subject: [PATCH 24/25] Studio: drop three comment blocks that later rounds made wrong The reserve constant carried two overlapping headers, the extras-drafter note still described the local-file-only rule that was reverted once remote repos could be priced, and the zero-bound note no longer matched the code. --- studio/backend/routes/inference.py | 17 ----------------- 1 file changed, 17 deletions(-) diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 60aef2f24e..65166bb8c1 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5674,11 +5674,6 @@ def _remote_gguf_companion_bytes( return 0 -# Stand-in for a drafter repo no listing could price. Matches the loader's own -# flat MTP reserve for the case where the drafter's dims are unavailable -# (llama_cpp's _tp_flat_mtp), because it answers the same question: the launch -# will make a drafter resident and nothing here can say how big it is. Zero is -# the one answer that is certainly wrong, since it admits the load. # What an unreadable remote drafter costs the guard. Sized to cover the largest # drafter class Studio knows of rather than a typical one: a DSpark sidecar is # about 11 GB (see the --fit note in llama_cpp._emit_dspark), --spec-draft-hf can @@ -5754,11 +5749,6 @@ def _remote_drafter_repo_bytes(spec: str, *, hf_token: Optional[str]) -> int: # the basename restores every quant and charges the repo's F16. matched = {n: s for n, s in sizes.items() if hint in n.lower()} sizes = matched or sizes - # An empty listing is "we learned nothing" and falls through to the cache - # and then the reserve. A listing that DID name GGUFs and still bounds at - # zero is different: every family was rejected as an incomplete split, so - # the fetch can load none of them and no draft weights become resident. - # Charging the reserve there 409s a load for VRAM nothing will take. bounded = dflash_budget_bytes(sizes, _gguf_extra_shards) if bounded: return bounded @@ -6037,13 +6027,6 @@ def _estimate_gguf_required_gb( # fits. Both halves matter. Owning the spec type alone keeps the conservative # charge, since the guard protects a running training job and a drafter can # still arrive by a route this cannot see. Applies to every kind, hence one flag. - # Only a LOCAL file, because the suppression's whole premise is that the - # drafter is "already charged below as _extras_bytes" and that charge is - # itself gated on Path(...).is_file(). --spec-draft-hf / -hfd names an HF - # repo id, which llama-server downloads and loads all the same (its - # has_dft() only asks whether a draft path was given), so suppressing on - # one would leave a multi-GB resident drafter charged nowhere and let the - # guard admit a load that evicts the training job it protects. _extras_own_draft_path = _extra_args_mtp_draft_path(llama_extra_args, env = {}) # An extras draft path wins whether or not the extras also own --spec-type: # the loader ranks _cli_draft_for_budget ahead of _studio_draft_for_budget, From 49b82bc7888c419ffba801945ebb7237d8567910 Mon Sep 17 00:00:00 2001 From: Daniel Han Date: Wed, 12 Aug 2026 12:52:30 +0000 Subject: [PATCH 25/25] Add staging CI workflows for unslothai/unsloth#8435 --- .github/workflows/staging-8435-macos-14.yml | 29 +++++++++++++++++++ .../workflows/staging-8435-ubuntu-latest.yml | 29 +++++++++++++++++++ .../workflows/staging-8435-windows-latest.yml | 29 +++++++++++++++++++ 3 files changed, 87 insertions(+) create mode 100644 .github/workflows/staging-8435-macos-14.yml create mode 100644 .github/workflows/staging-8435-ubuntu-latest.yml create mode 100644 .github/workflows/staging-8435-windows-latest.yml diff --git a/.github/workflows/staging-8435-macos-14.yml b/.github/workflows/staging-8435-macos-14.yml new file mode 100644 index 0000000000..c02ebbf4bf --- /dev/null +++ b/.github/workflows/staging-8435-macos-14.yml @@ -0,0 +1,29 @@ +name: "staging-8435 macos-14" +on: + push: + branches: ["pr-8435-xplat-ci"] + workflow_dispatch: +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true +permissions: + contents: read +defaults: + run: + shell: bash +jobs: + test: + runs-on: macos-14 + timeout-minutes: 30 + env: + UNSLOTH_COMPILE_DISABLE: '1' + UNSLOTH_IS_PRESENT: '1' + steps: + - uses: actions/checkout@v4 + with: + persist-credentials: false + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + - run: python -m pip install -r studio/backend/requirements/studio.txt pytest pytest-asyncio pytest-timeout python-multipart + - run: rc=0; PYTHONPATH=studio/backend python -m pytest studio/backend/tests/test_chat_load_during_training.py studio/backend/tests/test_compute_buffer.py studio/backend/tests/test_llama_cpp_placement.py studio/backend/tests/test_mtp_drafter_companion.py studio/backend/tests/test_mtp_vram_budget.py -q --tb=short -k 'not llama_cpp_load_progress_live and not TestGpuAutoSelection and not TestPreSpawnGpuResolution and not TestPerGpuFitGuardAllCounts and not TestTransformersIntrospection and not test_returns_cuda_when_cuda_available and not test_calls_cuda_cache_when_cuda' > pytest_out.txt 2>&1 || rc=$?; cat pytest_out.txt; if [ "$rc" = "5" ]; then echo "no tests ran (deps absent on this runner)"; rc=0; fi; if [ "$rc" = "2" ] && grep -qE "No module named .(torch|unsloth_zoo)." pytest_out.txt && ! grep -qE "^(FAILED|ERROR) " pytest_out.txt; then echo "collection needs a dep this runner does not ship"; rc=0; fi; exit "$rc" diff --git a/.github/workflows/staging-8435-ubuntu-latest.yml b/.github/workflows/staging-8435-ubuntu-latest.yml new file mode 100644 index 0000000000..29d0c92046 --- /dev/null +++ b/.github/workflows/staging-8435-ubuntu-latest.yml @@ -0,0 +1,29 @@ +name: "staging-8435 ubuntu-latest" +on: + push: + branches: ["pr-8435-xplat-ci"] + workflow_dispatch: +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true +permissions: + contents: read +defaults: + run: + shell: bash +jobs: + test: + runs-on: ubuntu-latest + timeout-minutes: 30 + env: + UNSLOTH_COMPILE_DISABLE: '1' + UNSLOTH_IS_PRESENT: '1' + steps: + - uses: actions/checkout@v4 + with: + persist-credentials: false + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + - run: python -m pip install -r studio/backend/requirements/studio.txt pytest pytest-asyncio pytest-timeout python-multipart + - run: rc=0; PYTHONPATH=studio/backend python -m pytest studio/backend/tests/test_chat_load_during_training.py studio/backend/tests/test_compute_buffer.py studio/backend/tests/test_llama_cpp_placement.py studio/backend/tests/test_mtp_drafter_companion.py studio/backend/tests/test_mtp_vram_budget.py -q --tb=short -k 'not llama_cpp_load_progress_live and not TestGpuAutoSelection and not TestPreSpawnGpuResolution and not TestPerGpuFitGuardAllCounts and not TestTransformersIntrospection and not test_returns_cuda_when_cuda_available and not test_calls_cuda_cache_when_cuda' > pytest_out.txt 2>&1 || rc=$?; cat pytest_out.txt; if [ "$rc" = "5" ]; then echo "no tests ran (deps absent on this runner)"; rc=0; fi; if [ "$rc" = "2" ] && grep -qE "No module named .(torch|unsloth_zoo)." pytest_out.txt && ! grep -qE "^(FAILED|ERROR) " pytest_out.txt; then echo "collection needs a dep this runner does not ship"; rc=0; fi; exit "$rc" diff --git a/.github/workflows/staging-8435-windows-latest.yml b/.github/workflows/staging-8435-windows-latest.yml new file mode 100644 index 0000000000..ef58bfecbc --- /dev/null +++ b/.github/workflows/staging-8435-windows-latest.yml @@ -0,0 +1,29 @@ +name: "staging-8435 windows-latest" +on: + push: + branches: ["pr-8435-xplat-ci"] + workflow_dispatch: +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true +permissions: + contents: read +defaults: + run: + shell: bash +jobs: + test: + runs-on: windows-latest + timeout-minutes: 30 + env: + UNSLOTH_COMPILE_DISABLE: '1' + UNSLOTH_IS_PRESENT: '1' + steps: + - uses: actions/checkout@v4 + with: + persist-credentials: false + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + - run: python -m pip install -r studio/backend/requirements/studio.txt pytest pytest-asyncio pytest-timeout python-multipart + - run: rc=0; PYTHONPATH=studio/backend python -m pytest studio/backend/tests/test_chat_load_during_training.py studio/backend/tests/test_compute_buffer.py studio/backend/tests/test_llama_cpp_placement.py studio/backend/tests/test_mtp_drafter_companion.py studio/backend/tests/test_mtp_vram_budget.py -q --tb=short -k 'not llama_cpp_load_progress_live and not TestGpuAutoSelection and not TestPreSpawnGpuResolution and not TestPerGpuFitGuardAllCounts and not TestTransformersIntrospection and not test_returns_cuda_when_cuda_available and not test_calls_cuda_cache_when_cuda' > pytest_out.txt 2>&1 || rc=$?; cat pytest_out.txt; if [ "$rc" = "5" ]; then echo "no tests ran (deps absent on this runner)"; rc=0; fi; if [ "$rc" = "2" ] && grep -qE "No module named .(torch|unsloth_zoo)." pytest_out.txt && ! grep -qE "^(FAILED|ERROR) " pytest_out.txt; then echo "collection needs a dep this runner does not ship"; rc=0; fi; exit "$rc"