From b965a285ca6f71d7d699225bd355562fb31f4b81 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 26 Jul 2026 04:45:34 +0000 Subject: [PATCH 01/56] Studio: say which model is missing instead of "No model loaded" A /v1 request naming a model that is not downloaded returned the generic "No model loaded. Call POST /inference/load first.", which cannot fix it. Return 404 model_not_found naming the model and listing what can serve, and make the API usage examples name a model the server actually has. --- .../core/inference/local_model_resolver.py | 27 ++ studio/backend/routes/inference.py | 174 +++++++++++-- .../backend/tests/test_openai_auto_switch.py | 241 +++++++++++++++++- .../features/settings/api/openai-models.ts | 38 +++ .../settings/components/usage-examples.tsx | 209 ++++----------- ...st_usage_examples_model_source_contract.py | 47 ++++ 6 files changed, 546 insertions(+), 190 deletions(-) create mode 100644 studio/frontend/src/features/settings/api/openai-models.ts create mode 100644 tests/studio/test_usage_examples_model_source_contract.py diff --git a/studio/backend/core/inference/local_model_resolver.py b/studio/backend/core/inference/local_model_resolver.py index 9e3eaeda3f8..3c5e55ed873 100644 --- a/studio/backend/core/inference/local_model_resolver.py +++ b/studio/backend/core/inference/local_model_resolver.py @@ -277,3 +277,30 @@ def resolve_local_gguf(requested: str) -> Optional[tuple[str, Optional[str], str # Best-effort: any resolver failure falls through to the loaded model, # so a malformed name can never turn a servable request into a 500. return None + + +MISS_MODEL_NOT_FOUND = "model_not_found" +MISS_VARIANT_NOT_FOUND = "variant_not_found" + + +def describe_local_miss(requested: str) -> tuple[str, tuple[str, ...]]: + """Why :func:`resolve_local_gguf` missed, so an error can say "wrong quant" + instead of "no such model". + + ``(MISS_VARIANT_NOT_FOUND, )`` when the repo is downloaded but + the requested ``:VARIANT`` is not, else ``(MISS_MODEL_NOT_FOUND, ())``. Splits + the name like the resolver so the two agree. Fail-safe: a scan failure reports + the generic miss rather than raising into the handler. + """ + if not isinstance(requested, str) or not requested.strip(): + return MISS_MODEL_NOT_FOUND, () + base, sep, _variant = requested.strip().rpartition(":") + if not sep: + return MISS_MODEL_NOT_FOUND, () + try: + entry = _index().get(base.strip().lower()) + except Exception: + return MISS_MODEL_NOT_FOUND, () + if entry is None or not entry.variants: + return MISS_MODEL_NOT_FOUND, () + return MISS_VARIANT_NOT_FOUND, entry.variants diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 9d40650a569..46df46a54a7 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -28,7 +28,7 @@ # Model size extraction (shared with core/inference/llama_cpp.py) from utils.models import extract_model_size_b as _extract_model_size_b -from utils.api_errors import openai_error_body, anthropic_error_body +from utils.api_errors import openai_error_body, anthropic_error_body, error_body_for_path from utils.upload_limits import STT_AUDIO_B64_MAX_CHARS, STT_AUDIO_RAW_MAX_BYTES from hub.dependencies import get_hf_token from core.inference.orchestrator import GenStreamError, GenStreamErrorRaised @@ -3562,6 +3562,116 @@ def _no_model_loaded_detail(base: str) -> str: ) +# Cap on ids listed by a "not downloaded" error: useful in a terminal without +# burying the message. +_MAX_LISTED_AVAILABLE_MODELS = 8 + + +def _raw_body_model(body) -> Optional[str]: + """The ``model`` a raw-body endpoint was given, else None (same value + :func:`_auto_switch_from_request_body` fed the switch hook).""" + return body.get("model") if isinstance(body, dict) else None + + +async def _available_model_ids() -> list[str]: + """Sorted ids a /v1 request may name, from the catalog ``GET /v1/models`` + serves, so an error and the listing can't disagree.""" + return sorted( + mid for mid in (m.get("id") for m in await _openai_catalog_objects()) + if isinstance(mid, str) and mid + ) + + +def _format_available_models(ids: list[str]) -> str: + if not ids: + return "" + shown = ", ".join(ids[:_MAX_LISTED_AVAILABLE_MODELS]) + extra = len(ids) - _MAX_LISTED_AVAILABLE_MODELS + return f"{shown} and {extra} more" if extra > 0 else shown + + +async def _unavailable_model_message(requested_model: str) -> str: + """Why a named model can't serve this request, and what can. + + Auto-switch only loads already-downloaded GGUFs, so a request naming a real + model usually fails because it is not on this machine. Pointing the caller at + /inference/load cannot fix that; say what is actually wrong. + """ + from core.inference.local_model_resolver import ( + MISS_VARIANT_NOT_FOUND, + describe_local_miss, + ) + + reason, variants = await asyncio.to_thread(describe_local_miss, requested_model) + if reason == MISS_VARIANT_NOT_FOUND: + # Repo downloaded, only the quant missing: sibling quants beat the catalog. + base_id, _, wanted = requested_model.strip().rpartition(":") + return ( + f"The model '{base_id}' is downloaded, but the quant '{wanted}' is not. " + f"Available quants: {', '.join(variants)}." + ) + available = _format_available_models(await _available_model_ids()) + if not available: + return ( + f"The model '{requested_model}' is not downloaded on this server, and no " + "models are downloaded yet. Download one in Unsloth Studio." + ) + return ( + f"The model '{requested_model}' is not downloaded on this server. " + f"Available models: {available}. Download more in Unsloth Studio, " + "or list them with GET /v1/models." + ) + + +async def _no_model_loaded_error( + base: str, + requested_model: Optional[str], + fastapi_request: Optional[Request], + *, + status: int, +): + """``(status, detail)`` for the /v1 sites that fail because nothing is loaded. + + Changes only the case the generic text describes wrongly: auto-switch on, a + model named, and that name resolves to nothing local, so the switch silently + did nothing. That becomes a 404 model_not_found. Toggle off or no model named + keeps ``status`` and the :func:`_no_model_loaded_detail` text verbatim. + """ + from utils.openai_auto_switch_settings import get_openai_auto_switch_enabled + from core.inference.local_model_resolver import resolve_local_gguf + + named = ( + requested_model + if isinstance(requested_model, str) + and requested_model.strip() + and requested_model != _RELOAD_ONLY_MODEL + else None + ) + if named is None or not get_openai_auto_switch_enabled(): + return status, _no_model_loaded_detail(base) + try: + if await asyncio.to_thread(resolve_local_gguf, named) is not None: + # Resolvable but unloaded: the switch failed (load error, race), which + # the generic text describes correctly. + return status, _no_model_loaded_detail(base) + message = await _unavailable_model_message(named) + except Exception as exc: + # The diagnosis is a nicety; never let it turn a 4xx into a 500. + logger.debug("no-model-loaded diagnosis failed for %r: %s", named, exc) + return status, _no_model_loaded_detail(base) + path = getattr(getattr(fastapi_request, "url", None), "path", None) + if not isinstance(path, str): + # No request in hand: let the global /v1/* handler pick the envelope. + return 404, message + return 404, error_body_for_path( + path, + message, + status = 404, + code = "model_not_found", + param = "model", + ) + + async def _maybe_auto_switch_model( requested_model: Optional[str], fastapi_request: Request, @@ -7754,10 +7864,13 @@ async def _monitored_generate_audio(model_label: str, context_length: Optional[i else: backend = get_inference_backend() if not backend.active_model_name: - raise HTTPException( - status_code = 400, - detail = _no_model_loaded_detail("No model loaded. Call POST /inference/load first."), + _status, _detail = await _no_model_loaded_error( + "No model loaded. Call POST /inference/load first.", + _switch_model_for_payload(payload), + request, + status = 400, ) + raise HTTPException(status_code = _status, detail = _detail) # Clean public id so the response never echoes a local path; the audio # branch below receives this sanitized label too. model_name = public_model_id(backend.active_model_name) or payload.model @@ -10721,10 +10834,13 @@ async def openai_completions(request: Request, current_subject: str = Depends(ge # Opt-in: load the requested local GGUF before the loaded-state check. body = await _auto_switch_from_request_body(request, current_subject) if not llama_backend.is_loaded: - raise HTTPException( - status_code = 503, - detail = _no_model_loaded_detail("No GGUF model loaded. Load a GGUF model first."), + _status, _detail = await _no_model_loaded_error( + "No GGUF model loaded. Load a GGUF model first.", + _raw_body_model(body), + request, + status = 503, ) + raise HTTPException(status_code = _status, detail = _detail) if not isinstance(body, dict): # Re-read to re-raise a malformed-body error (post-503, pre-feature behavior); # a valid non-dict body such as a list is a clean 400 rather than a 500. @@ -10941,10 +11057,13 @@ async def openai_embeddings(request: Request, current_subject: str = Depends(get # a non-embedding target switches, then llama-server returns a no-pooling error. body = await _auto_switch_from_request_body(request, current_subject) if not llama_backend.is_loaded: - raise HTTPException( - status_code = 503, - detail = _no_model_loaded_detail("No GGUF model loaded. Load a GGUF model first."), + _status, _detail = await _no_model_loaded_error( + "No GGUF model loaded. Load a GGUF model first.", + _raw_body_model(body), + request, + status = 503, ) + raise HTTPException(status_code = _status, detail = _detail) if not isinstance(body, dict): # Re-read to re-raise a malformed-body error (post-503, pre-feature behavior); # a valid non-dict body such as a list is a clean 400 rather than a 500. @@ -11681,14 +11800,15 @@ async def _responses_stream( # double-layer asyncgen close pattern that produces "Attempted to exit # cancel scope in a different task" on Python 3.13. Surface a typed 400 # so the client sees a useful error instead of a dangling stream. - raise HTTPException( - status_code = 400, - detail = _no_model_loaded_detail( - "Streaming /v1/responses requires a GGUF model loaded via " - "llama-server. Use non-streaming /v1/responses, " - "/v1/chat/completions, or load a GGUF model." - ), + _status, _detail = await _no_model_loaded_error( + "Streaming /v1/responses requires a GGUF model loaded via " + "llama-server. Use non-streaming /v1/responses, " + "/v1/chat/completions, or load a GGUF model.", + _switch_model_for_payload(payload), + request, + status = 400, ) + raise HTTPException(status_code = _status, detail = _detail) # Direct pass-through bypasses the openai_chat_completions image gate. if not llama_backend.is_vision and any( @@ -12930,10 +13050,13 @@ async def anthropic_count_tokens( llama_backend = get_llama_cpp_backend() if not llama_backend.is_loaded: - raise HTTPException( - status_code = 503, - detail = _no_model_loaded_detail("No GGUF model loaded. Load a GGUF model first."), + _status, _detail = await _no_model_loaded_error( + "No GGUF model loaded. Load a GGUF model first.", + _switch_model_for_payload(payload), + request, + status = 503, ) + raise HTTPException(status_code = _status, detail = _detail) # Same Anthropic → OpenAI translation as anthropic_messages: system is # folded into the messages list, so pass system=None to the counter. @@ -13002,6 +13125,8 @@ async def anthropic_messages( # before any request-shape check, exactly as the pre-feature endpoint did. When # an automatic load can run (auto-switch or a standalone idle TTL), fall through # so validation runs before the reload hook gets a chance to restore the model. + # Plain detail, not _no_model_loaded_error: no automatic load was possible, the + # case that helper leaves unchanged. The 404 comes from the post-switch check below. if not llama_backend.is_loaded and not _automatic_model_load_may_run(): raise HTTPException( status_code = 503, @@ -13101,10 +13226,13 @@ async def anthropic_messages( require_vision = _anthropic_request_has_image(payload), ) if not llama_backend.is_loaded: - raise HTTPException( - status_code = 503, - detail = _no_model_loaded_detail("No GGUF model loaded. Load a GGUF model first."), + _status, _detail = await _no_model_loaded_error( + "No GGUF model loaded. Load a GGUF model first.", + _switch_model_for_payload(payload), + request, + status = 503, ) + raise HTTPException(status_code = _status, detail = _detail) # Advertised repo id after an auto-switch load, else a clean public id, never # the local .gguf path (and a legacy raw path in payload.model is sanitized). diff --git a/studio/backend/tests/test_openai_auto_switch.py b/studio/backend/tests/test_openai_auto_switch.py index 9c6c20e6b6f..a614a0bc030 100644 --- a/studio/backend/tests/test_openai_auto_switch.py +++ b/studio/backend/tests/test_openai_auto_switch.py @@ -387,6 +387,50 @@ def test_resolver_nonstring_model_is_failsafe(): assert resolver.resolve_local_gguf(None) is None +def test_describe_local_miss_separates_missing_repo_from_missing_quant(monkeypatch): + # A miss is two different situations and the error should say which: the repo + # isn't downloaded, or it is and only that quant is absent. + monkeypatch.setattr( + resolver, + "_build_index", + lambda: {"unsloth/b-gguf": _entry("unsloth/B-GGUF", "UD-Q5_K_XL", "Q4_K_M")}, + ) + resolver._scan = (0.0, {}) + assert resolver.describe_local_miss("unsloth/B-GGUF:Q8_0") == ( + resolver.MISS_VARIANT_NOT_FOUND, + ("UD-Q5_K_XL", "Q4_K_M"), + ) + # Split the same way resolve_local_gguf does, so the two never disagree. + assert resolver.describe_local_miss("unsloth/b-gguf:q8_0")[0] == ( + resolver.MISS_VARIANT_NOT_FOUND + ) + # Unknown repo, and a bare id with no ":VARIANT" to blame. + assert resolver.describe_local_miss("totally/unknown:Q8_0") == ( + resolver.MISS_MODEL_NOT_FOUND, + (), + ) + assert resolver.describe_local_miss("unsloth/B-GGUF") == ( + resolver.MISS_MODEL_NOT_FOUND, + (), + ) + + +def test_describe_local_miss_is_failsafe(monkeypatch): + # Runs inside an error path, so a broken scan must degrade to the generic miss + # rather than raise a 500 over what was already a 4xx. + def boom(): + raise RuntimeError("scan blew up") + + monkeypatch.setattr(resolver, "_build_index", boom) + resolver._scan = (0.0, {}) + assert resolver.describe_local_miss("unsloth/B-GGUF:Q8_0") == ( + resolver.MISS_MODEL_NOT_FOUND, + (), + ) + assert resolver.describe_local_miss(123) == (resolver.MISS_MODEL_NOT_FOUND, ()) + assert resolver.describe_local_miss("") == (resolver.MISS_MODEL_NOT_FOUND, ()) + + def test_resolver_exact_id_with_colon_wins(monkeypatch): # A local id that itself contains a colon (e.g. a Windows path) must match # exactly rather than being split at the drive-letter colon. @@ -3240,13 +3284,14 @@ def test_no_model_loaded_detail_appends_hint_only_when_off(monkeypatch): assert inference_route._no_model_loaded_detail(base) == base -def _run_responses_stream_no_model(monkeypatch, *, enabled, active_model_name): +def _run_responses_stream_no_model(monkeypatch, *, enabled, active_model_name, resolves_to = None): # Drive _responses_stream's GGUF-not-loaded guard: llama backend unloaded, - # inference backend maybe holding a non-GGUF model. Returns the 400 detail. + # inference backend maybe holding a non-GGUF model. Returns (status, detail). from fastapi import HTTPException from models.inference import ResponsesRequest, ChatMessage monkeypatch.setattr(settings, "get_openai_auto_switch_enabled", lambda: enabled) + monkeypatch.setattr(resolver, "resolve_local_gguf", lambda name: resolves_to) monkeypatch.setattr( inference_route, "get_llama_cpp_backend", lambda: _FakeBackend(loaded_id = None) ) @@ -3259,29 +3304,199 @@ def _run_responses_stream_no_model(monkeypatch, *, enabled, active_model_name): messages = [ChatMessage(role = "user", content = "hi")] with pytest.raises(HTTPException) as exc: asyncio.run(inference_route._responses_stream(payload, messages, None)) - assert exc.value.status_code == 400 - return exc.value.detail + return exc.value.status_code, exc.value.detail def test_responses_stream_hint_matches_toggle_regardless_of_active_model(monkeypatch): - # Streaming /v1/responses shares the GGUF-only 400 with the other "no model - # loaded" sites, so the auto-switch hint attaches whenever the toggle is - # off -- including while a non-GGUF model is active, since auto-switch - # evicts it to load a resolved GGUF (_maybe_auto_switch_model's resolver - # branch has no active-model guard, unlike its reload-stash branch). Only - # the toggle being on suppresses it. - hinted = _run_responses_stream_no_model(monkeypatch, enabled = False, active_model_name = None) + # The hint attaches whenever the toggle is off, including while a non-GGUF + # model is active, since auto-switch evicts it to load a resolved GGUF + # (the resolver branch has no active-model guard, unlike the reload stash). + # With it on the error changes kind: the name resolved to nothing local, so + # 404 model_not_found rather than a 400 pointing at /inference/load. + off_status, hinted = _run_responses_stream_no_model( + monkeypatch, enabled = False, active_model_name = None + ) + assert off_status == 400 assert "Model auto-switch" in hinted - on = _run_responses_stream_no_model(monkeypatch, enabled = True, active_model_name = None) + on_status, on = _run_responses_stream_no_model( + monkeypatch, enabled = True, active_model_name = None + ) + assert on_status == 404 assert "Model auto-switch" not in on + assert "unsloth/Qwen3.5-4B-GGUF" in on - non_gguf_loaded = _run_responses_stream_no_model( + non_gguf_status, non_gguf_loaded = _run_responses_stream_no_model( monkeypatch, enabled = False, active_model_name = "unsloth/Llama-3.2-1B-Instruct" ) + assert non_gguf_status == 400 assert "Model auto-switch" in non_gguf_loaded +def _wire_unloaded_chat(monkeypatch, *, enabled, catalog = ("org/A-GGUF", "org/B-GGUF")): + # Nothing loaded, so a chat request reaches the "no model loaded" error. Pin + # the catalog so listed ids don't depend on what is cached on the test machine. + async def _catalog(): + return [{"id": mid} for mid in catalog] + + monkeypatch.setattr(settings, "get_openai_auto_switch_enabled", lambda: enabled) + monkeypatch.setattr(resolver, "resolve_local_gguf", lambda _m: None) + monkeypatch.setattr( + resolver, "describe_local_miss", lambda _m: (resolver.MISS_MODEL_NOT_FOUND, ()) + ) + monkeypatch.setattr(inference_route, "_openai_catalog_objects", _catalog) + monkeypatch.setattr( + inference_route, "get_llama_cpp_backend", lambda: _FakeBackend(loaded_id = None) + ) + monkeypatch.setattr( + inference_route, + "get_inference_backend", + lambda: type("_B", (), {"active_model_name": None, "models": {}})(), + ) + + +def _chat_error(payload): + from fastapi import HTTPException + + with pytest.raises(HTTPException) as exc: + asyncio.run(inference_route.openai_chat_completions(payload, object(), "tester")) + return exc.value.status_code, exc.value.detail + + +def test_chat_names_undownloaded_model_404s_with_available_ids(monkeypatch): + # The reported bug: auto-switch on, the named model is not on this machine, so + # the switch silently did nothing and /inference/load cannot fix it. Name the + # model and list what can serve, from the catalog GET /v1/models returns. + _wire_unloaded_chat(monkeypatch, enabled = True) + status, detail = _chat_error(_chat_request(model = "unsloth/gemma-4-E4B-it-GGUF:UD-Q5_K_XL")) + assert status == 404 + assert "unsloth/gemma-4-E4B-it-GGUF:UD-Q5_K_XL" in detail + assert "org/A-GGUF, org/B-GGUF" in detail + assert "GET /v1/models" in detail + assert "POST /inference/load" not in detail + + +def test_chat_undownloaded_model_with_empty_catalog(monkeypatch): + # Nothing downloaded: an empty list would read as a bug, so say so plainly. + _wire_unloaded_chat(monkeypatch, enabled = True, catalog = ()) + status, detail = _chat_error(_chat_request(model = "org/nope-GGUF")) + assert status == 404 + assert "no models are downloaded yet" in detail + + +def test_chat_wrong_quant_lists_the_local_quants(monkeypatch): + # Repo downloaded, only the quant missing: sibling quants, not the catalog. + _wire_unloaded_chat(monkeypatch, enabled = True) + monkeypatch.setattr( + resolver, + "describe_local_miss", + lambda _m: (resolver.MISS_VARIANT_NOT_FOUND, ("Q4_K_M", "Q8_0")), + ) + status, detail = _chat_error(_chat_request(model = "org/A-GGUF:UD-Q5_K_XL")) + assert status == 404 + assert "'org/A-GGUF' is downloaded, but the quant 'UD-Q5_K_XL' is not" in detail + assert "Q4_K_M, Q8_0" in detail + + +def test_chat_error_unchanged_when_auto_switch_off(monkeypatch): + # Toggle off: nothing was resolved, so "not downloaded" would be a guess. Keep + # the pre-existing status and text, hint included. + _wire_unloaded_chat(monkeypatch, enabled = False) + status, detail = _chat_error(_chat_request(model = "org/nope-GGUF")) + assert status == 400 + assert detail.startswith("No model loaded. Call POST /inference/load first.") + assert "Model auto-switch" in detail + + +def test_chat_error_unchanged_when_no_model_named(monkeypatch): + # An omitted model means "serve whatever is loaded", so the generic text is + # right and there is no name to report as missing. + _wire_unloaded_chat(monkeypatch, enabled = True) + status, detail = _chat_error(_chat_request()) + assert status == 400 + assert detail == "No model loaded. Call POST /inference/load first." + + +def test_chat_not_downloaded_error_survives_a_broken_catalog_scan(monkeypatch): + # The diagnosis is layered onto an already-failing path: a scan blowing up + # must not turn the 4xx into a 500. + async def _boom(): + raise RuntimeError("catalog scan blew up") + + _wire_unloaded_chat(monkeypatch, enabled = True) + monkeypatch.setattr(inference_route, "_openai_catalog_objects", _boom) + status, detail = _chat_error(_chat_request(model = "org/nope-GGUF")) + assert status == 400 + assert detail.startswith("No model loaded. Call POST /inference/load first.") + + +def test_chat_available_id_list_is_capped(monkeypatch): + # A machine with 40 GGUFs must not print all 40 into a terminal error. + _wire_unloaded_chat( + monkeypatch, enabled = True, catalog = tuple(f"org/m{i:02d}-GGUF" for i in range(20)) + ) + status, detail = _chat_error(_chat_request(model = "org/nope-GGUF")) + assert status == 404 + assert "and 12 more" in detail + assert "org/m08-GGUF" not in detail + + +def test_anthropic_undownloaded_model_uses_the_anthropic_envelope(monkeypatch): + # Shared with /v1/messages, so the 404 must not leak an OpenAI-shaped body. + from fastapi import HTTPException + + async def _noop_switch(*a, **k): + return None + + _wire_unloaded_chat(monkeypatch, enabled = True) + monkeypatch.setattr(inference_route, "_automatic_model_load_may_run", lambda: True) + monkeypatch.setattr(inference_route, "_maybe_auto_switch_model", _noop_switch) + + request = type("_R", (), {"url": type("_U", (), {"path": "/v1/messages"})()})() + with pytest.raises(HTTPException) as exc: + asyncio.run( + inference_route.anthropic_messages(_anthropic_payload(64), request, "tester") + ) + assert exc.value.status_code == 404 + body = exc.value.detail + assert body["type"] == "error" + assert body["error"]["type"] == "not_found_error" + assert "claude-x" in body["error"]["message"] + + +def test_chat_undownloaded_model_uses_the_openai_envelope(monkeypatch): + # On the OpenAI surface it carries param/code so SDK clients can branch on it, + # matching GET /v1/models/{id}'s existing model_not_found. + from fastapi import HTTPException + + _wire_unloaded_chat(monkeypatch, enabled = True) + request = type("_R", (), {"url": type("_U", (), {"path": "/v1/chat/completions"})()})() + with pytest.raises(HTTPException) as exc: + asyncio.run( + inference_route.openai_chat_completions( + _chat_request(model = "org/nope-GGUF"), request, "tester" + ) + ) + assert exc.value.status_code == 404 + err = exc.value.detail["error"] + assert err["type"] == "not_found_error" + assert err["code"] == "model_not_found" + assert err["param"] == "model" + + +def test_responses_stream_keeps_generic_error_when_target_is_local(monkeypatch): + # Name resolves locally yet nothing is loaded: the switch failed, so the + # generic text is right and the status stays 400, not a bogus "missing". + status, detail = _run_responses_stream_no_model( + monkeypatch, + enabled = True, + active_model_name = None, + resolves_to = ("/p/A", "Q4_K_M", "unsloth/Qwen3.5-4B-GGUF"), + ) + assert status == 400 + assert "not downloaded" not in detail + + # ── idle-unload KV persistence (slot save/restore) ────────────────── diff --git a/studio/frontend/src/features/settings/api/openai-models.ts b/studio/frontend/src/features/settings/api/openai-models.ts new file mode 100644 index 00000000000..31c651f5c2d --- /dev/null +++ b/studio/frontend/src/features/settings/api/openai-models.ts @@ -0,0 +1,38 @@ +// SPDX-License-Identifier: AGPL-3.0-only +// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0 + +import { authFetch } from "@/features/auth"; + +export type OpenAIModel = { + id: string; + // Resident in memory now; the rest are downloaded and servable. + loaded?: boolean; +}; + +type ApiOpenAIModelList = { + data?: { id?: unknown; loaded?: unknown }[]; +}; + +/** + * The models this server can serve. + * + * `/v1/models` rather than the hub inventory: it returns exactly the ids + * `/v1/chat/completions` resolves against, so a snippet built from it is + * runnable as printed. `/v1` accepts the UI session JWT as well as + * `sk-unsloth-` keys, so `authFetch` works unchanged. + */ +export async function listOpenAIModels(): Promise { + const res = await authFetch("/v1/models"); + if (!res.ok) { + throw new Error(`Failed to list models (${res.status})`); + } + const body = (await res.json()) as ApiOpenAIModelList; + if (!Array.isArray(body?.data)) { + return []; + } + return body.data.flatMap((entry) => + typeof entry?.id === "string" && entry.id + ? [{ id: entry.id, loaded: entry.loaded === true }] + : [], + ); +} diff --git a/studio/frontend/src/features/settings/components/usage-examples.tsx b/studio/frontend/src/features/settings/components/usage-examples.tsx index ade181f632b..3fdb65ea3a2 100644 --- a/studio/frontend/src/features/settings/components/usage-examples.tsx +++ b/studio/frontend/src/features/settings/components/usage-examples.tsx @@ -29,11 +29,7 @@ import { HugeiconsIcon } from "@hugeicons/react"; import { useEffect, useMemo, useRef, useState } from "react"; import { Streamdown } from "streamdown"; import { loadCodingAgents } from "../api/coding-agents"; -import { - type OpenAIAutoSwitchSettings, - loadOpenAIAutoSwitchSettings, - updateOpenAIAutoSwitchSettings, -} from "../api/openai-auto-switch"; +import { type OpenAIModel, listOpenAIModels } from "../api/openai-models"; import { buildAgentCommand, isLoopbackHost, normalizeHost } from "./agent-command"; type ExampleType = @@ -90,13 +86,6 @@ const JAVASCRIPT_TYPES = new Set([ ]); const PROMPT = "Can Unsloth Studio do API calling?"; -// Auto-switch demo: a second call naming a different downloaded GGUF so the -// example shows that the model field selects which model serves. -// A placeholder the user replaces with one of their downloaded GGUFs. A fixed -// repo is usually not one they have, so the resolver would fall through and the -// demo would keep serving the current model instead of switching. -const SWITCH_MODEL = "your-other-downloaded-GGUF"; -const SWITCH_PROMPT = "Now answer as a different model."; // web_search + python + terminal are the reliable built-in tools. const TOOLS = ["web_search", "python", "terminal"]; const ADV = { @@ -195,19 +184,13 @@ function winBody(model: string, variant: Variant): string { return JSON.stringify(body, null, 2); } -// A leading comment (valid in both bash and PowerShell) noting the model field -// selects the served model when auto-switch is on. -const SWITCH_NOTE = - '# "Switch model by request" is on: set "model" to any downloaded GGUF to switch.\n'; - function curlUnix( base: string, key: string, model: string, variant: Variant, - autoSwitch: boolean, ): string { - return `${autoSwitch ? SWITCH_NOTE : ""}curl ${base}/v1/chat/completions \\ + return `curl ${base}/v1/chat/completions \\ -H "Authorization: Bearer ${key}" \\ -H "Content-Type: application/json" \\ -d '${shSingle(curlBodyPretty(model, variant))}'`; @@ -218,9 +201,8 @@ function curlWindows( key: string, model: string, variant: Variant, - autoSwitch: boolean, ): string { - return `${autoSwitch ? SWITCH_NOTE : ""}$body = '${psSingle(winBody(model, variant))}' + return `$body = '${psSingle(winBody(model, variant))}' Set-Content -Path body.json -Value $body -Encoding ascii curl.exe ${base}/v1/chat/completions \` -H "Authorization: Bearer ${key}" \` @@ -228,29 +210,11 @@ curl.exe ${base}/v1/chat/completions \` -d "@body.json"`; } -// A second OpenAI call naming a different downloaded GGUF: with auto-switch on, -// Unsloth loads it before serving, so the model field selects the served model. -function pythonSwitchDemo(): string { - return ` - -# "Switch model by request" is on: replace the model below with another GGUF you -# have downloaded and Unsloth loads it before serving. Unknown names keep serving -# the current model. -response = client.chat.completions.create( - model=${j(SWITCH_MODEL)}, - messages=[{"role": "user", "content": ${j(SWITCH_PROMPT)}}], - stream=True, -) -for chunk in response: - print(chunk.choices[0].delta.content or "", end="")`; -} - function pythonSnippet( base: string, key: string, model: string, variant: Variant, - autoSwitch: boolean, ): string { const named = variant === "advanced" @@ -295,7 +259,7 @@ response = client.chat.completions.create( messages=[{"role": "user", "content": ${j(PROMPT)}}],${named}${extraBody} stream=True, ) -${loop}${autoSwitch ? pythonSwitchDemo() : ""}`; +${loop}`; } function javascriptSnippet( @@ -303,7 +267,6 @@ function javascriptSnippet( key: string, model: string, variant: Variant, - autoSwitch: boolean, ): string { const options: string[] = []; if (variant === "advanced") { @@ -342,23 +305,6 @@ const response = await client.chat.completions.create({ for await (const chunk of response) { process.stdout.write(chunk.choices?.[0]?.delta?.content || ""); -}${autoSwitch ? javascriptSwitchDemo() : ""}`; -} - -function javascriptSwitchDemo(): string { - return ` - -// "Switch model by request" is on: replace the model below with another GGUF you -// have downloaded and Unsloth loads it before serving. Unknown names keep serving -// the current model. -const switchResponse = await client.chat.completions.create({ - model: ${j(SWITCH_MODEL)}, - messages: [{ role: "user", content: ${j(SWITCH_PROMPT)} }], - stream: true, -}); - -for await (const chunk of switchResponse) { - process.stdout.write(chunk.choices?.[0]?.delta?.content || ""); }`; } @@ -367,25 +313,18 @@ function buildSnippets( key: string, model: string, os: Os, - autoSwitch: boolean, ): Record { const curl = os === "windows" ? curlWindows : curlUnix; return { - curl: curl(base, key, model, "plain", autoSwitch), - python: pythonSnippet(base, key, model, "plain", autoSwitch), - javascript: javascriptSnippet(base, key, model, "plain", autoSwitch), - curlTools: curl(base, key, model, "tools", autoSwitch), - pythonTools: pythonSnippet(base, key, model, "tools", autoSwitch), - javascriptTools: javascriptSnippet(base, key, model, "tools", autoSwitch), - curlAdvanced: curl(base, key, model, "advanced", autoSwitch), - pythonAdvanced: pythonSnippet(base, key, model, "advanced", autoSwitch), - javascriptAdvanced: javascriptSnippet( - base, - key, - model, - "advanced", - autoSwitch, - ), + curl: curl(base, key, model, "plain"), + python: pythonSnippet(base, key, model, "plain"), + javascript: javascriptSnippet(base, key, model, "plain"), + curlTools: curl(base, key, model, "tools"), + pythonTools: pythonSnippet(base, key, model, "tools"), + javascriptTools: javascriptSnippet(base, key, model, "tools"), + curlAdvanced: curl(base, key, model, "advanced"), + pythonAdvanced: pythonSnippet(base, key, model, "advanced"), + javascriptAdvanced: javascriptSnippet(base, key, model, "advanced"), }; } @@ -411,18 +350,46 @@ function writeUseTunnelPref(value: boolean): void { } } -function useLoadedModelName(): string { +// The model the examples name. With nothing loaded this used to print +// MODEL_FALLBACK, a repo id the user likely never downloaded, so the copied +// snippet 404d. Prefer a real id from the catalog /v1 resolves against; the +// constant is only the floor for a server with no models, so it is never blank. +function useExampleModelName(): string { const checkpoint = useChatRuntimeStore((s) => s.params.checkpoint); const ggufVariant = useChatRuntimeStore((s) => s.activeGgufVariant); + const [catalog, setCatalog] = useState([]); + const needsCatalog = !checkpoint || checkpoint.startsWith("external::"); + + // Only when the checkpoint can't answer it: /v1/models scans the model dirs and + // HF caches. Re-runs when a model is loaded or unloaded. + useEffect(() => { + if (!needsCatalog) return; + let cancelled = false; + void listOpenAIModels() + .then((models) => { + if (!cancelled) setCatalog(models); + }) + .catch(() => { + // Best-effort: fall through to the placeholder. + }); + return () => { + cancelled = true; + }; + }, [needsCatalog]); + return useMemo(() => { - if (!checkpoint || checkpoint.startsWith("external::")) { - return MODEL_FALLBACK; - } - if (ggufVariant && !checkpoint.includes(":")) { - return `${checkpoint}:${ggufVariant}`; + if (checkpoint && !checkpoint.startsWith("external::")) { + if (ggufVariant && !checkpoint.includes(":")) { + return `${checkpoint}:${ggufVariant}`; + } + return checkpoint; } - return checkpoint; - }, [checkpoint, ggufVariant]); + // No local checkpoint (or an external provider, which /v1 cannot serve): + // name something this server actually holds. + return ( + catalog.find((m) => m.loaded)?.id ?? catalog[0]?.id ?? MODEL_FALLBACK + ); + }, [catalog, checkpoint, ggufVariant]); } // Backend PATH detection is only safe in the desktop app, where the UI owns @@ -492,11 +459,6 @@ export function UsageExamples({ apiKey }: { apiKey?: string | null }) { const base = useTunnel && cloudflareUrl ? cloudflareUrl : (serverUrl ?? origin); const localAgentDetection = canUseLocalAgentDetection(base); - // null while loading; the same setting the General tab exposes (shared cache). - const [autoSwitch, setAutoSwitch] = useState( - null, - ); - const [savingAutoSwitch, setSavingAutoSwitch] = useState(false); useEffect(() => { void fetchDeviceType({ force: true }); @@ -574,27 +536,12 @@ export function UsageExamples({ apiKey }: { apiKey?: string | null }) { } }, [agent, detectedAgents, activeGgufVariant, activeNativePathToken, ggufContextLength]); - useEffect(() => { - let cancelled = false; - void loadOpenAIAutoSwitchSettings() - .then((s) => { - if (!cancelled) setAutoSwitch(s); - }) - .catch(() => { - // Best-effort: leave the toggle off if the setting can't be read. - }); - return () => { - cancelled = true; - }; - }, []); - - const model = useLoadedModelName(); + const model = useExampleModelName(); const key = apiKey || KEY_PLACEHOLDER; - const autoSwitchOn = autoSwitch?.enabled ?? false; const snippets = useMemo( - () => buildSnippets(base, key, model, os, autoSwitchOn), - [base, key, model, os, autoSwitchOn], + () => buildSnippets(base, key, model, os), + [base, key, model, os], ); // Agent command must target the server the panel shows, not the :8888 default. const agentCommand = useMemo( @@ -623,20 +570,6 @@ export function UsageExamples({ apiKey }: { apiKey?: string | null }) { writeUseTunnelPref(next); }; - // Same setting as the General tab; persist optimistically and revert on failure - // so the examples reflect the live model-switch behavior. - const handleToggleAutoSwitch = (next: boolean) => { - const idle = autoSwitch?.autoUnloadIdleSeconds ?? 0; - setAutoSwitch((prev) => (prev ? { ...prev, enabled: next } : prev)); - setSavingAutoSwitch(true); - void updateOpenAIAutoSwitchSettings(next, idle) - .then(setAutoSwitch) - .catch(() => { - setAutoSwitch((prev) => (prev ? { ...prev, enabled: !next } : prev)); - }) - .finally(() => setSavingAutoSwitch(false)); - }; - const handleCopyUrl = async () => { if (cloudflareUrl && (await copyToClipboard(cloudflareUrl))) { setCopiedUrl(true); @@ -657,41 +590,9 @@ export function UsageExamples({ apiKey }: { apiKey?: string | null }) { {t("settings.apiKeys.usageExamples")}
- {/* Same setting as the General tab; surfaced here so the request `model` - actually switches the served model, which the examples below show. */} -
-
- - - {t("settings.general.modelAutoSwitch.enable")} - - - - - - - {t("settings.general.modelAutoSwitch.enableDescription")} - - -
-
+ {/* No model-auto-switch row: ModelAutoSwitchSection renders the same + setting just below on this tab, and the two do not share state (the + settings module caches on first read), so a duplicate drifts. */} {cloudflareUrl ? (
diff --git a/tests/studio/test_usage_examples_model_source_contract.py b/tests/studio/test_usage_examples_model_source_contract.py new file mode 100644 index 00000000000..549957ea60c --- /dev/null +++ b/tests/studio/test_usage_examples_model_source_contract.py @@ -0,0 +1,47 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0 + +"""Static contract for which model the API usage examples name, and for the +model-auto-switch control living in exactly one place on the API keys tab.""" + +from pathlib import Path + +REPO = Path(__file__).resolve().parents[2] +SETTINGS = REPO / "studio/frontend/src/features/settings" +USAGE_EXAMPLES_TSX = SETTINGS / "components/usage-examples.tsx" +OPENAI_MODELS_TS = SETTINGS / "api/openai-models.ts" +API_KEYS_TAB_TSX = SETTINGS / "tabs/api-keys-tab.tsx" + + +def test_examples_name_a_model_the_server_can_serve(): + # The snippets used to fall back to a hardcoded repo id whenever nothing was + # loaded, so a copied curl named a model the user had never downloaded and + # 404d. Read the servable ids from /v1/models instead -- the same list the + # backend's model-not-downloaded error lists, so the two cannot disagree. + src = USAGE_EXAMPLES_TSX.read_text(encoding = "utf-8") + assert 'from "../api/openai-models"' in src + assert "function useExampleModelName(): string" in src + hook = src[src.find("function useExampleModelName") : src.find("// Backend PATH detection")] + assert "listOpenAIModels()" in hook + # Precedence: live checkpoint, then a loaded catalog entry, then any entry, + # and only then the placeholder. + assert "catalog.find((m) => m.loaded)?.id ?? catalog[0]?.id ?? MODEL_FALLBACK" in hook + + api = OPENAI_MODELS_TS.read_text(encoding = "utf-8") + assert 'authFetch("/v1/models")' in api + + +def test_usage_examples_has_no_duplicate_auto_switch_control(): + # ModelAutoSwitchSection renders the same setting immediately below this + # panel on the same tab, and the two do not share state, so a second switch + # here drifts out of sync with the one the user can also see. + src = USAGE_EXAMPLES_TSX.read_text(encoding = "utf-8") + assert "openai-auto-switch" not in src + assert "SWITCH_NOTE" not in src + assert "Switch model by request" not in src + assert "pythonSwitchDemo" not in src + assert "javascriptSwitchDemo" not in src + assert "modelAutoSwitch" not in src + + tab = API_KEYS_TAB_TSX.read_text(encoding = "utf-8") + assert "" in tab From fed31afa45d1be720c9f4c603d49a9b6bcdb5a44 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Sun, 26 Jul 2026 04:46:49 +0000 Subject: [PATCH 02/56] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/routes/inference.py | 9 +++---- .../backend/tests/test_openai_auto_switch.py | 25 +++++++++++-------- 2 files changed, 18 insertions(+), 16 deletions(-) diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 46df46a54a7..289dd268682 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -3577,7 +3577,8 @@ async def _available_model_ids() -> list[str]: """Sorted ids a /v1 request may name, from the catalog ``GET /v1/models`` serves, so an error and the listing can't disagree.""" return sorted( - mid for mid in (m.get("id") for m in await _openai_catalog_objects()) + mid + for mid in (m.get("id") for m in await _openai_catalog_objects()) if isinstance(mid, str) and mid ) @@ -3624,11 +3625,7 @@ async def _unavailable_model_message(requested_model: str) -> str: async def _no_model_loaded_error( - base: str, - requested_model: Optional[str], - fastapi_request: Optional[Request], - *, - status: int, + base: str, requested_model: Optional[str], fastapi_request: Optional[Request], *, status: int ): """``(status, detail)`` for the /v1 sites that fail because nothing is loaded. diff --git a/studio/backend/tests/test_openai_auto_switch.py b/studio/backend/tests/test_openai_auto_switch.py index a614a0bc030..a3c96ec23c5 100644 --- a/studio/backend/tests/test_openai_auto_switch.py +++ b/studio/backend/tests/test_openai_auto_switch.py @@ -409,10 +409,7 @@ def test_describe_local_miss_separates_missing_repo_from_missing_quant(monkeypat resolver.MISS_MODEL_NOT_FOUND, (), ) - assert resolver.describe_local_miss("unsloth/B-GGUF") == ( - resolver.MISS_MODEL_NOT_FOUND, - (), - ) + assert resolver.describe_local_miss("unsloth/B-GGUF") == (resolver.MISS_MODEL_NOT_FOUND, ()) def test_describe_local_miss_is_failsafe(monkeypatch): @@ -3284,7 +3281,13 @@ def test_no_model_loaded_detail_appends_hint_only_when_off(monkeypatch): assert inference_route._no_model_loaded_detail(base) == base -def _run_responses_stream_no_model(monkeypatch, *, enabled, active_model_name, resolves_to = None): +def _run_responses_stream_no_model( + monkeypatch, + *, + enabled, + active_model_name, + resolves_to = None, +): # Drive _responses_stream's GGUF-not-loaded guard: llama backend unloaded, # inference backend maybe holding a non-GGUF model. Returns (status, detail). from fastapi import HTTPException @@ -3333,7 +3336,12 @@ def test_responses_stream_hint_matches_toggle_regardless_of_active_model(monkeyp assert "Model auto-switch" in non_gguf_loaded -def _wire_unloaded_chat(monkeypatch, *, enabled, catalog = ("org/A-GGUF", "org/B-GGUF")): +def _wire_unloaded_chat( + monkeypatch, + *, + enabled, + catalog = ("org/A-GGUF", "org/B-GGUF"), +): # Nothing loaded, so a chat request reaches the "no model loaded" error. Pin # the catalog so listed ids don't depend on what is cached on the test machine. async def _catalog(): @@ -3357,7 +3365,6 @@ async def _catalog(): def _chat_error(payload): from fastapi import HTTPException - with pytest.raises(HTTPException) as exc: asyncio.run(inference_route.openai_chat_completions(payload, object(), "tester")) return exc.value.status_code, exc.value.detail @@ -3454,9 +3461,7 @@ async def _noop_switch(*a, **k): request = type("_R", (), {"url": type("_U", (), {"path": "/v1/messages"})()})() with pytest.raises(HTTPException) as exc: - asyncio.run( - inference_route.anthropic_messages(_anthropic_payload(64), request, "tester") - ) + asyncio.run(inference_route.anthropic_messages(_anthropic_payload(64), request, "tester")) assert exc.value.status_code == 404 body = exc.value.detail assert body["type"] == "error" From 88f1e5eb4a8a3db9ef95b1466d191b6b223e51b9 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 26 Jul 2026 07:31:35 +0000 Subject: [PATCH 03/56] Studio: page the API monitor, show model load/unload, pin the example quant The monitor rendered all 50 retained entries in one scroller: page it 5 at a time, freezing history while paged back so live traffic cannot reorder it. Add model load/unload rows so the feed shows what the server is doing, and stop the header rendering the loaded model as a raw host path. Advertise each model's GGUF quant on /v1/models so the example pins repo:QUANT, and move the auto-switch section above the monitor with shorter copy. --- studio/backend/core/inference/api_monitor.py | 95 ++++++++++- .../backend/core/inference/llama_keepwarm.py | 19 +++ .../core/inference/local_model_resolver.py | 22 ++- studio/backend/routes/inference.py | 92 ++++++++++- studio/backend/tests/test_api_monitor.py | 101 ++++++++++++ studio/backend/tests/test_openai_catalog.py | 54 +++++- .../frontend/src/features/chat/types/api.ts | 5 + .../features/settings/api/openai-models.ts | 13 +- .../components/api-monitor-console.tsx | 154 ++++++++++++++++-- .../settings/components/usage-examples.tsx | 37 ++++- .../features/settings/tabs/api-keys-tab.tsx | 4 +- studio/frontend/src/i18n/locales/en.ts | 12 +- ...st_usage_examples_model_source_contract.py | 35 +++- 13 files changed, 591 insertions(+), 52 deletions(-) diff --git a/studio/backend/core/inference/api_monitor.py b/studio/backend/core/inference/api_monitor.py index f76a38576fc..2390d235a2e 100644 --- a/studio/backend/core/inference/api_monitor.py +++ b/studio/backend/core/inference/api_monitor.py @@ -52,6 +52,13 @@ class ApiMonitorEntry: total_tokens: Optional[int] = None total_tokens_authoritative: bool = False error: Optional[str] = None + # "request" (an HTTP call) or "lifecycle" (a model load/unload). Lifecycle + # rows carry event/reason instead of a prompt and are server-wide, so they + # are shared across subjects rather than owned by the caller that caused them. + kind: str = "request" + event: Optional[str] = None + reason: Optional[str] = None + shared: bool = False def snapshot(self, *, include_details: bool = True) -> dict[str, Any]: duration_ms = None @@ -85,6 +92,9 @@ def snapshot(self, *, include_details: bool = True) -> dict[str, Any]: "completion_tokens": self.completion_tokens, "total_tokens": self.total_tokens, "error": self.error, + "kind": self.kind, + "event": self.event, + "reason": self.reason, } if include_details: payload["prompt"] = self.prompt @@ -127,6 +137,65 @@ def start( self._trim_terminal_locked() return entry.id + def record_lifecycle( + self, + *, + event: str, + model: str, + reason: Optional[str] = None, + running: bool = False, + ) -> str: + """Record a model load/unload alongside the request traffic that caused it. + + ``running=True`` opens the row (a load in progress) and the caller closes + it with the usual :meth:`finish` / :meth:`fail`; an unload is terminal on + arrival. Rows are shared, so every subject sees them, and share the same + retention budget as requests. + """ + now = time.time() + entry = ApiMonitorEntry( + id = f"apievt_{uuid.uuid4().hex[:12]}", + endpoint = f"model.{event}", + method = "", + model = model or "default", + prompt = "", + status = "running" if running else "completed", + started_at = now, + updated_at = now, + started_monotonic = time.monotonic(), + finished_at = None if running else now, + finished_monotonic = None if running else time.monotonic(), + kind = "lifecycle", + event = event, + reason = reason, + shared = True, + ) + with self._lock: + self._entries.appendleft(entry) + self._trim_terminal_locked() + return entry.id + + def relabel(self, entry_id: Optional[str], model: str) -> None: + """Rename an open lifecycle row once the load resolves its real id (the + caller only has the load path up front, which may be an HF snapshot dir).""" + if not entry_id or not model: + return + with self._lock: + entry = self._find_locked(entry_id) + if entry is not None: + entry.model = model + entry.updated_at = time.time() + + def discard(self, entry_id: Optional[str]) -> None: + """Drop a row that turned out not to be an event (a load that was already + satisfied, so nothing was actually loaded).""" + if not entry_id: + return + with self._lock: + entry = self._find_locked(entry_id) + if entry is not None: + self._entries.remove(entry) + def append_reply(self, entry_id: Optional[str], text: str) -> None: if not entry_id or not text: return @@ -212,6 +281,18 @@ def finish( self._entries.appendleft(entry) self._trim_terminal_locked() + def fail_open(self, entry_id: Optional[str], error: str) -> None: + """Fail only a still-open row. Unlike :meth:`fail` this never touches an + entry that already finished, so a catch-all in a ``finally`` cannot stamp + an error onto a request that in fact succeeded.""" + if not entry_id: + return + with self._lock: + entry = self._find_locked(entry_id) + if entry is None or entry.finished_at is not None: + return + self.fail(entry_id, error) + def fail(self, entry_id: Optional[str], error: str) -> None: if not entry_id: return @@ -244,7 +325,7 @@ def snapshot( return [ entry.snapshot(include_details = include_details) for entry in self._entries - if subject is None or entry.subject == subject + if self._visible(entry, subject) ] def get( @@ -257,22 +338,30 @@ def get( entry = self._find_locked(entry_id) if entry is None: return None - if subject is not None and entry.subject != subject: + if not self._visible(entry, subject): return None return entry.snapshot(include_details = True) def active_count(self, *, subject: Optional[str] = None) -> int: + # Lifecycle rows are excluded: a load in progress is "running" so the row + # can show as live, but it is not an in-flight API request. with self._lock: return sum( 1 for entry in self._entries - if entry.status == "running" and (subject is None or entry.subject == subject) + if entry.status == "running" + and entry.kind != "lifecycle" + and (subject is None or entry.subject == subject) ) def clear(self) -> None: with self._lock: self._entries.clear() + @staticmethod + def _visible(entry: ApiMonitorEntry, subject: Optional[str]) -> bool: + return subject is None or entry.subject == subject or entry.shared + def _find_locked(self, entry_id: str) -> Optional[ApiMonitorEntry]: for entry in self._entries: if entry.id == entry_id: diff --git a/studio/backend/core/inference/llama_keepwarm.py b/studio/backend/core/inference/llama_keepwarm.py index 3380ebf5f59..44ce54d8ba5 100644 --- a/studio/backend/core/inference/llama_keepwarm.py +++ b/studio/backend/core/inference/llama_keepwarm.py @@ -345,6 +345,22 @@ def _loaded_identity(backend): return (backend.model_identifier, getattr(backend, "hf_variant", None), advertised) +def _note_idle_unload_event(freed) -> None: + """Record an idle auto-unload in the API monitor, using the advertised repo id + from the stash so the row never shows the on-disk load path. Best-effort.""" + try: + from core.inference.api_monitor import api_monitor + from core.inference.model_ids import public_model_id + + identifier, variant, advertised = (list(freed) + [None, None, None])[:3] + label = public_model_id(advertised or identifier) or "model" + if variant and ":" not in label: + label = f"{label}:{variant}" + api_monitor.record_lifecycle(event = "unload", model = label, reason = "idle") + except Exception as exc: + logger.debug("idle unload monitor event failed: %s", exc) + + async def idle_unload_loop(poll_seconds: float = 15.0) -> None: """Unload the loaded GGUF once idle past the configured TTL. Inert when off.""" from utils.openai_auto_switch_settings import ( @@ -407,6 +423,9 @@ async def idle_unload_loop(poll_seconds: float = 15.0) -> None: elif manifest: _delete_resume_files(manifest) logger.info("Idle auto-unload: freed GGUF after %ss idle", ttl) + # This path deliberately skips note_model_unloaded (an idle + # unload stashes for reload), so record the monitor row here. + _note_idle_unload_event(freed) seen_model = None except Exception as exc: logger.debug("idle_unload_loop iteration failed: %s", exc) diff --git a/studio/backend/core/inference/local_model_resolver.py b/studio/backend/core/inference/local_model_resolver.py index 3c5e55ed873..0533db944f6 100644 --- a/studio/backend/core/inference/local_model_resolver.py +++ b/studio/backend/core/inference/local_model_resolver.py @@ -108,12 +108,12 @@ def _local_gguf_entry(loader_id: str, info) -> Optional[_LocalGgufEntry]: return None -def info_has_local_gguf(info) -> bool: - """True when *info* (a LocalModelInfo) points to on-disk GGUF weights the - auto-switch path can load. Read from the files, not ``info.model_format``: the - HF-cache scanner leaves model_format unset for GGUF snapshots, so a - model_format filter would drop every cached GGUF. Lets /v1/models advertise - exactly what /v1 can serve.""" +def local_gguf_quants(info) -> Optional[tuple[str, ...]]: + """On-disk quant labels for *info*, or None when it is not a servable local + GGUF. Read from the files, not ``info.model_format``: the HF-cache scanner + leaves model_format unset for GGUF snapshots, so a model_format filter would + drop every cached GGUF. Lets /v1/models advertise exactly what /v1 can serve, + and which quant to name, from a single scan.""" from pathlib import Path path = getattr(info, "path", None) @@ -123,8 +123,14 @@ def info_has_local_gguf(info) -> bool: if isinstance(path, str) and any( seg in (".studio_links", "ollama_links") for seg in Path(path).parts ): - return False - return _local_gguf_entry(getattr(info, "id", "") or "", info) is not None + return None + entry = _local_gguf_entry(getattr(info, "id", "") or "", info) + return entry.variants if entry is not None else None + + +def info_has_local_gguf(info) -> bool: + """True when *info* points to on-disk GGUF weights the auto-switch path can load.""" + return local_gguf_quants(info) is not None def _build_index() -> dict[str, _LocalGgufEntry]: diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 289dd268682..dc63eff45ea 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -3070,12 +3070,50 @@ def _monitor_context_length() -> Optional[int]: return None +def _hf_cache_repo_id(path: Optional[str]) -> Optional[str]: + """``.../models--org--name/snapshots/`` -> ``org/name``, else None. + + An auto-switch load is handed the concrete snapshot dir, whose basename is the + commit sha, so public_model_id alone would label the row with a hash. + """ + if not path: + return None + for part in str(path).replace("\\", "/").split("/"): + if part.startswith("models--"): + return part[len("models--") :].replace("--", "/") + return None + + +def _lifecycle_model_label(model: Optional[str], variant: Optional[str] = None) -> str: + """A path-free ``repo`` / ``repo:QUANT`` label for a monitor lifecycle row.""" + clean = _hf_cache_repo_id(model) or public_model_id(model) or model or "model" + return f"{clean}:{variant}" if variant and ":" not in clean else clean + + +def _close_load_event( + entry_id: Optional[str], model: Optional[str], variant: Optional[str] +) -> None: + """Close a monitor load row, relabelled with the id the load actually resolved + (the row opened on the request's model_path, which may be an HF snapshot dir).""" + api_monitor.relabel(entry_id, _lifecycle_model_label(model, variant)) + api_monitor.finish(entry_id) + + def _monitor_active_model() -> Optional[str]: + """The loaded model as a client-facing id, quant included when known. + + Cleaned like /v1/models: this is rendered in the settings UI and served over + the public --secure tunnel, so it must never be the on-disk load path. + """ llama_backend = get_llama_cpp_backend() if getattr(llama_backend, "is_loaded", False): - return getattr(llama_backend, "model_identifier", None) + model_id = _llama_public_model_id(llama_backend) + variant = getattr(llama_backend, "hf_variant", None) + if model_id and variant and ":" not in model_id: + return f"{model_id}:{variant}" + return model_id backend = get_inference_backend() - return backend.active_model_name + return public_model_id(backend.active_model_name) or backend.active_model_name def _validate_native_gguf_companion( @@ -4402,6 +4440,15 @@ async def _load_model_impl( # sampled step logs even if it reports 100% immediately (cached/small load). _reset_load_progress_step() + # Open a live "loading" row in the API monitor. Discarded below if the model + # turned out to be already loaded, relabelled once the real id is known, and + # closed on every exit by the handlers at the end of this function. + _load_event = api_monitor.record_lifecycle( + event = "load", + model = _lifecycle_model_label(request.model_path, request.gguf_variant), + running = True, + ) + native_grant_backed = False model_log_label = request.model_path gguf_load_stack = ExitStack() @@ -4495,6 +4542,8 @@ async def _load_model_impl( and getattr(llama_backend, "_audio_probed", True) ): llama_backend._record_matching_gpu_request(request.gpu_ids) + # Nothing was loaded, so the monitor must not show a load row. + api_monitor.discard(_load_event) logger.info( "Model already loaded (GGUF): " f"{model_log_label} variant={request.gguf_variant or llama_backend.hf_variant}, skipping reload" @@ -4548,6 +4597,7 @@ async def _load_model_impl( backend.active_model_name and backend.active_model_name.lower() == model_identifier.lower() ): + api_monitor.discard(_load_event) # nothing loaded, no monitor row logger.info(f"Model already loaded (Unsloth): {model_log_label}, skipping reload") inference_config = load_inference_config(backend.active_model_name) _model_info = backend.models.get(backend.active_model_name, {}) @@ -4860,6 +4910,11 @@ async def _attempt_gguf_load( logger.info( f"Loaded GGUF model via llama-server: {model_log_label if native_grant_backed else config.identifier}" ) + _close_load_event( + _load_event, + model_log_label if native_grant_backed else config.identifier, + request.gguf_variant or getattr(llama_backend, "hf_variant", None), + ) # Clear any idle-unload reload stash now, not only on the next poll. from core.inference.llama_keepwarm import note_model_loaded @@ -4978,6 +5033,9 @@ async def _attempt_gguf_load( logger.info( f"Loaded model: {model_log_label if native_grant_backed else config.identifier}" ) + _close_load_event( + _load_event, model_log_label if native_grant_backed else config.identifier, None + ) # Clear any idle-unload reload stash: a manual load supersedes an idle-freed # GGUF, so the next /v1 request must not resurrect it. Mirror the GGUF branch # above; without this a non-GGUF load leaves a stale stash until the idle @@ -5100,6 +5158,9 @@ async def _attempt_gguf_load( raise HTTPException(status_code = 500, detail = f"Failed to load model: {msg}") finally: gguf_load_stack.close() + # Catch-all: any exit that did not already close or discard the row (an + # error, or a cancelled load) leaves it stuck "loading" otherwise. + api_monitor.fail_open(_load_event, "Load did not complete") def _requires_trust_remote_code_for_model( @@ -5720,6 +5781,11 @@ async def unload_model(request: UnloadRequest, current_subject: str = Depends(ge # request is mid-stream (only the automatic idle loop defers to it). llama_backend.unload_model() note_model_unloaded() + api_monitor.record_lifecycle( + event = "unload", + model = _lifecycle_model_label(request.model_path), + reason = "manual", + ) logger.info(f"Unloaded GGUF model: {request.model_path}") return UnloadResponse(status = "unloaded", model = request.model_path) @@ -5729,6 +5795,11 @@ async def unload_model(request: UnloadRequest, current_subject: str = Depends(ge backend = get_inference_backend() await asyncio.to_thread(backend.unload_model, request.model_path) note_model_unloaded() + api_monitor.record_lifecycle( + event = "unload", + model = _lifecycle_model_label(request.model_path), + reason = "manual", + ) logger.info(f"Unloaded model: {request.model_path}") return UnloadResponse(status = "unloaded", model = request.model_path) @@ -10575,6 +10646,9 @@ def _openai_model_objects() -> list[dict]: "created": _created, "owned_by": _OWNED_BY, } + _quant = getattr(llama_backend, "hf_variant", None) + if _quant: + entry["quant"] = _quant _ctx = _positive_int_or_none(getattr(llama_backend, "context_length", None)) if _ctx is not None: entry["context_length"] = _ctx @@ -10681,11 +10755,15 @@ async def _openai_catalog_objects() -> list[dict]: # read from the on-disk files, not model_format: the HF-cache scanner leaves # model_format unset for GGUF snapshots, so a model_format filter would drop # every cached GGUF. The file checks run off the loop. - from core.inference.local_model_resolver import info_has_local_gguf + from core.inference.local_model_resolver import local_gguf_quants catalog = await _cached_local_catalog() - servable = await asyncio.to_thread(lambda: [i for i in catalog if info_has_local_gguf(i)]) - for info in servable: + # One scan yields both "is this servable" and its on-disk quants, so an entry + # can advertise the quant to name without a second pass over the model dirs. + servable = await asyncio.to_thread( + lambda: [(i, q) for i in catalog if (q := local_gguf_quants(i)) is not None] + ) + for info, quants in servable: cid = getattr(info, "model_id", None) or public_model_id(getattr(info, "id", None)) if not cid or cid in by_id: continue @@ -10696,6 +10774,10 @@ async def _openai_catalog_objects() -> list[dict]: "owned_by": _OWNED_BY, "loaded": False, } + # Bare label (never a filename or path): the id stays bare for OpenAI + # compat, and a client appends ":" to pin this exact quant. + if quants: + obj["quant"] = quants[0] display = getattr(info, "display_name", None) if display: obj["display_name"] = display diff --git a/studio/backend/tests/test_api_monitor.py b/studio/backend/tests/test_api_monitor.py index 56bc4043503..44b1f5fecde 100644 --- a/studio/backend/tests/test_api_monitor.py +++ b/studio/backend/tests/test_api_monitor.py @@ -258,3 +258,104 @@ def test_api_monitor_append_reply_exact_cap_then_more_marks_truncated(): monitor.append_reply(entry_id, "y") reply = monitor.snapshot()[0]["reply"] assert len(reply) == m._MAX_REPLY_CHARS and reply.endswith("...") + + +# ── model lifecycle rows (load / unload) ──────────────────────────── + + +def test_lifecycle_load_row_opens_running_then_closes(): + monitor = ApiMonitor(max_entries = 5) + event_id = monitor.record_lifecycle(event = "load", model = "org/A-GGUF", running = True) + row = monitor.snapshot()[0] + assert row["kind"] == "lifecycle" and row["event"] == "load" + assert row["status"] == "running" and row["duration_ms"] is None + # A load in progress is not an in-flight API request. + assert monitor.active_count() == 0 + + monitor.relabel(event_id, "org/A-GGUF:Q4_K_M") + monitor.finish(event_id) + row = monitor.snapshot()[0] + assert row["status"] == "completed" + assert row["model"] == "org/A-GGUF:Q4_K_M" + assert row["duration_ms"] is not None + + +def test_lifecycle_unload_row_is_terminal_on_arrival(): + monitor = ApiMonitor(max_entries = 5) + monitor.record_lifecycle(event = "unload", model = "org/A-GGUF", reason = "idle") + row = monitor.snapshot()[0] + assert row["status"] == "completed" + assert (row["event"], row["reason"]) == ("unload", "idle") + assert monitor.active_count() == 0 + + +def test_lifecycle_rows_are_visible_to_every_subject(): + # A load is server-wide, not owned by whoever happened to trigger it, so it + # must not vanish for other API keys the way a request does. + monitor = ApiMonitor(max_entries = 5) + monitor.start( + endpoint = "/v1/chat/completions", + method = "POST", + model = "m", + prompt = "hi", + subject = "alice", + ) + event_id = monitor.record_lifecycle(event = "unload", model = "org/A-GGUF") + + bob = monitor.snapshot(subject = "bob") + assert [r["kind"] for r in bob] == ["lifecycle"] + assert monitor.get(event_id, subject = "bob") is not None + assert len(monitor.snapshot(subject = "alice")) == 2 + + +def test_request_rows_stay_private_to_their_subject(): + monitor = ApiMonitor(max_entries = 5) + rid = monitor.start( + endpoint = "/v1/chat/completions", + method = "POST", + model = "m", + prompt = "hi", + subject = "alice", + ) + assert monitor.snapshot(subject = "bob") == [] + assert monitor.get(rid, subject = "bob") is None + + +def test_discard_drops_a_row_that_never_happened(): + # A load that found the model already resident must leave no trace. + monitor = ApiMonitor(max_entries = 5) + event_id = monitor.record_lifecycle(event = "load", model = "org/A-GGUF", running = True) + monitor.discard(event_id) + assert monitor.snapshot() == [] + monitor.discard(event_id) # idempotent + + +def test_fail_open_never_touches_a_finished_row(): + # The load path calls this from a finally, so it must not stamp an error onto + # a load that in fact succeeded. + monitor = ApiMonitor(max_entries = 5) + event_id = monitor.record_lifecycle(event = "load", model = "org/A-GGUF", running = True) + monitor.finish(event_id) + monitor.fail_open(event_id, "Load did not complete") + row = monitor.snapshot()[0] + assert row["status"] == "completed" and row["error"] is None + + still_open = monitor.record_lifecycle(event = "load", model = "org/B-GGUF", running = True) + monitor.fail_open(still_open, "Load did not complete") + assert monitor.snapshot()[0]["status"] == "error" + + +def test_lifecycle_rows_share_the_retention_budget(): + monitor = ApiMonitor(max_entries = 2) + for i in range(4): + monitor.record_lifecycle(event = "unload", model = f"org/M{i}") + models = [r["model"] for r in monitor.snapshot()] + assert models == ["org/M3", "org/M2"] + + +def test_request_rows_report_kind_request(): + monitor = ApiMonitor(max_entries = 2) + monitor.start( + endpoint = "/v1/chat/completions", method = "POST", model = "m", prompt = "hi" + ) + assert monitor.snapshot()[0]["kind"] == "request" diff --git a/studio/backend/tests/test_openai_catalog.py b/studio/backend/tests/test_openai_catalog.py index 552f122ebb6..3c5c953b962 100644 --- a/studio/backend/tests/test_openai_catalog.py +++ b/studio/backend/tests/test_openai_catalog.py @@ -64,8 +64,11 @@ async def _fake_catalog(): ] monkeypatch.setattr(inf, "_cached_local_catalog", _fake_catalog) - # GGUF-ness is read from the on-disk files; drive it off each info's flag here. - monkeypatch.setattr(resolver, "info_has_local_gguf", lambda info: info.is_gguf) + # GGUF-ness and the quant labels both come from the on-disk files in one scan; + # drive them off each info's flag here. + monkeypatch.setattr( + resolver, "local_gguf_quants", lambda info: ("Q8_0",) if info.is_gguf else None + ) data = asyncio.run(inf._openai_catalog_objects()) ids = {m["id"]: m for m in data} @@ -73,8 +76,10 @@ async def _fake_catalog(): # Loaded model is present, marked loaded, and keeps context fields. assert ids["Qwen3-Q4"]["loaded"] is True assert ids["Qwen3-Q4"]["context_length"] == 4096 - # Available-but-not-loaded GGUF models are listed too. + # Available-but-not-loaded GGUF models are listed too, with the quant a client + # appends to the id to pin it. assert ids["Llama-8B-Q8"]["loaded"] is False + assert ids["Llama-8B-Q8"]["quant"] == "Q8_0" # The HF-cache GGUF is listed despite model_format being unset. assert ids["org/Foo"]["loaded"] is False # The non-GGUF model is filtered out (/v1 can never serve it). @@ -205,3 +210,46 @@ async def _run(): assert second is first or [i.id for i in second] == [i.id for i in first] assert calls["scan"] == 1 # cached: scanned once for two calls assert calls["threaded"] == 1 # offloaded to a worker thread + + +def test_monitor_active_model_is_a_public_id_not_a_host_path(monkeypatch): + # The settings UI renders this and --secure serves it over a public tunnel, so + # it must never be the on-disk load path (it used to be model_identifier raw). + class _Llama: + is_loaded = True + model_identifier = "/home/me/.cache/huggingface/hub/models--org--A-GGUF/snapshots/abc" + hf_variant = "UD-Q4_K_XL" + _openai_advertised_id = "org/A-GGUF" + + monkeypatch.setattr(inf, "get_llama_cpp_backend", lambda: _Llama()) + assert inf._monitor_active_model() == "org/A-GGUF:UD-Q4_K_XL" + + +def test_monitor_active_model_cleans_a_path_with_no_advertised_id(monkeypatch): + class _Llama: + is_loaded = True + model_identifier = "/data/models/Llama-8B-Q8.gguf" + hf_variant = None + _openai_advertised_id = None + + monkeypatch.setattr(inf, "get_llama_cpp_backend", lambda: _Llama()) + label = inf._monitor_active_model() + assert "/" not in label and ".gguf" not in label + + +def test_lifecycle_label_recovers_the_repo_id_from_an_hf_cache_path(): + # An auto-switch load is handed the snapshot dir, whose basename is a commit + # sha, so the row would otherwise be labelled with a hash. + snap = "/home/me/.cache/huggingface/hub/models--unsloth--gemma-4-E4B-it-GGUF/snapshots/bfc15c3" + assert ( + inf._lifecycle_model_label(snap, "UD-Q4_K_XL") + == "unsloth/gemma-4-E4B-it-GGUF:UD-Q4_K_XL" + ) + + +def test_lifecycle_model_label_is_path_free(): + label = inf._lifecycle_model_label("/data/models/Llama-8B-Q8.gguf", "Q8_0") + assert "/" not in label and ".gguf" not in label + assert inf._lifecycle_model_label("org/A-GGUF", "Q4_K_M") == "org/A-GGUF:Q4_K_M" + # An id that already carries a quant is not double-suffixed. + assert inf._lifecycle_model_label("org/A-GGUF:Q4_K_M", "Q8_0") == "org/A-GGUF:Q4_K_M" diff --git a/studio/frontend/src/features/chat/types/api.ts b/studio/frontend/src/features/chat/types/api.ts index 6c3e919efe7..7f50fcfdaf4 100644 --- a/studio/frontend/src/features/chat/types/api.ts +++ b/studio/frontend/src/features/chat/types/api.ts @@ -283,6 +283,11 @@ export interface ApiMonitorEntry { completion_tokens?: number | null; total_tokens?: number | null; error?: string | null; + // "request" is an HTTP call; "lifecycle" is a model load/unload, which carries + // event/reason instead of a prompt and has no detail to fetch. + kind?: "request" | "lifecycle"; + event?: "load" | "unload" | null; + reason?: "manual" | "idle" | null; } export interface ApiMonitorResponse { diff --git a/studio/frontend/src/features/settings/api/openai-models.ts b/studio/frontend/src/features/settings/api/openai-models.ts index 31c651f5c2d..d4f7c18a63d 100644 --- a/studio/frontend/src/features/settings/api/openai-models.ts +++ b/studio/frontend/src/features/settings/api/openai-models.ts @@ -7,10 +7,13 @@ export type OpenAIModel = { id: string; // Resident in memory now; the rest are downloaded and servable. loaded?: boolean; + // On-disk GGUF quant. Ids stay bare for OpenAI compat, so append it as + // `id:quant` to pin this exact quant rather than letting the server pick. + quant?: string; }; type ApiOpenAIModelList = { - data?: { id?: unknown; loaded?: unknown }[]; + data?: { id?: unknown; loaded?: unknown; quant?: unknown }[]; }; /** @@ -32,7 +35,13 @@ export async function listOpenAIModels(): Promise { } return body.data.flatMap((entry) => typeof entry?.id === "string" && entry.id - ? [{ id: entry.id, loaded: entry.loaded === true }] + ? [ + { + id: entry.id, + loaded: entry.loaded === true, + quant: typeof entry.quant === "string" ? entry.quant : undefined, + }, + ] : [], ); } diff --git a/studio/frontend/src/features/settings/components/api-monitor-console.tsx b/studio/frontend/src/features/settings/components/api-monitor-console.tsx index 0f931ff4daa..41cd504f47f 100644 --- a/studio/frontend/src/features/settings/components/api-monitor-console.tsx +++ b/studio/frontend/src/features/settings/components/api-monitor-console.tsx @@ -22,6 +22,7 @@ import type { ApiMonitorEntry, ApiMonitorResponse } from "../../chat/types/api"; const API_INFERENCE_PREFIX_RE = /^\/api\/inference/; const V1_PREFIX_RE = /^\/v1\//; +const PAGE_SIZE = 5; function formatTime(value: number): string { return new Date(value * 1000).toLocaleTimeString([], { @@ -87,6 +88,53 @@ function UsageBar({ value }: { value?: number | null }): ReactElement | null { ); } +function isLifecycle(entry: ApiMonitorEntry): boolean { + return entry.kind === "lifecycle"; +} + +function lifecycleLabel(entry: ApiMonitorEntry): string { + if (entry.event === "unload") { + return entry.reason === "idle" ? "Model unloaded (idle)" : "Model unloaded"; + } + if (entry.status === "running") { + return "Loading model"; + } + if (entry.status === "completed") { + return "Model loaded"; + } + return "Model load failed"; +} + +// Load/unload rows: a label, the model and a time. No prompt, reply, tokens or +// detail fetch, so unlike a request row there is nothing to expand. +function LifecycleEntry({ entry }: { entry: ApiMonitorEntry }): ReactElement { + return ( +
+
+
+
+ + + {lifecycleLabel(entry)} + +
+
+ {entry.model} +
+
+
+
{formatTime(entry.started_at)}
+ {entry.event === "load" ? ( +
{formatDuration(entry.duration_ms)}
+ ) : null} +
+
+
+ ); +} + function MonitorEntry({ entry, detail, @@ -253,6 +301,51 @@ export function ApiMonitorConsole(): ReactElement { const statusLabel = data?.status ?? "idle"; const hasActive = (data?.active_requests ?? 0) > 0; const entries = useMemo(() => data?.entries ?? [], [data]); + + // Page 1 tracks the live list. Paging back freezes the id order captured at + // that moment: the poll keeps refreshing row contents, but new traffic must not + // shove history down a page while it is being read (finishing a request also + // moves it to the head, so even an idle server reorders). + const [page, setPage] = useState(0); + const [frozenIds, setFrozenIds] = useState(null); + const byId = useMemo( + () => new Map(entries.map((entry) => [entry.id, entry])), + [entries], + ); + const ordered = useMemo(() => { + if (frozenIds === null) { + return entries; + } + return frozenIds.flatMap((id) => { + const entry = byId.get(id); + return entry ? [entry] : []; + }); + }, [byId, entries, frozenIds]); + const pageCount = Math.max(1, Math.ceil(ordered.length / PAGE_SIZE)); + const pageIndex = Math.min(page, pageCount - 1); + const visible = ordered.slice( + pageIndex * PAGE_SIZE, + pageIndex * PAGE_SIZE + PAGE_SIZE, + ); + const newerCount = + frozenIds === null + ? 0 + : entries.filter((entry) => !frozenIds.includes(entry.id)).length; + + const goToPage = useCallback( + (next: number): void => { + if (next <= 0) { + setFrozenIds(null); + setPage(0); + return; + } + // Freeze on the way off page 1 so the history under the cursor holds still. + setFrozenIds((prev) => prev ?? entries.map((entry) => entry.id)); + setPage(next); + }, + [entries], + ); + const loadDetail = useCallback( (id: string): void => { if (loadingDetailsRef.current.has(id)) { @@ -305,8 +398,10 @@ export function ApiMonitorConsole(): ReactElement { ); useEffect(() => { - for (const entry of entries) { - if (!expandedIds.has(entry.id)) { + // Only rows actually on screen: an expanded row left behind on another page + // would otherwise keep polling its detail every tick. + for (const entry of visible) { + if (isLifecycle(entry) || !expandedIds.has(entry.id)) { continue; } const cached = detailsRef.current[entry.id]; @@ -314,7 +409,7 @@ export function ApiMonitorConsole(): ReactElement { loadDetail(entry.id); } } - }, [entries, expandedIds, loadDetail]); + }, [visible, expandedIds, loadDetail]); return (
@@ -375,19 +470,52 @@ export function ApiMonitorConsole(): ReactElement {
) : (
- {entries.map((entry) => ( - toggleEntry(entry)} - /> - ))} + {visible.map((entry) => + isLifecycle(entry) ? ( + + ) : ( + toggleEntry(entry)} + /> + ), + )}
)}
+ + {ordered.length > PAGE_SIZE ? ( +
+ + Page {pageIndex + 1} of {pageCount} + {newerCount > 0 ? ` (${newerCount.toLocaleString()} new)` : ""} + +
+ + +
+
+ ) : null} ); } diff --git a/studio/frontend/src/features/settings/components/usage-examples.tsx b/studio/frontend/src/features/settings/components/usage-examples.tsx index 3fdb65ea3a2..e26a6ed4f63 100644 --- a/studio/frontend/src/features/settings/components/usage-examples.tsx +++ b/studio/frontend/src/features/settings/components/usage-examples.tsx @@ -350,6 +350,20 @@ function writeUseTunnelPref(value: boolean): void { } } +// A checkpoint can be an on-disk load path (an auto-switch load resolves to the +// GGUF/snapshot dir), which must never reach a printed snippet: it leaks the host +// layout and is not an id /v1 advertises. Mirrors the backend's _looks_like_path. +function looksLikePath(id: string): boolean { + return ( + id.startsWith("/") || + id.startsWith("~") || + id.startsWith(".") || + id.includes("\\") || + id.toLowerCase().endsWith(".gguf") || + (id.match(/\//g)?.length ?? 0) >= 2 + ); +} + // The model the examples name. With nothing loaded this used to print // MODEL_FALLBACK, a repo id the user likely never downloaded, so the copied // snippet 404d. Prefer a real id from the catalog /v1 resolves against; the @@ -358,7 +372,9 @@ function useExampleModelName(): string { const checkpoint = useChatRuntimeStore((s) => s.params.checkpoint); const ggufVariant = useChatRuntimeStore((s) => s.activeGgufVariant); const [catalog, setCatalog] = useState([]); - const needsCatalog = !checkpoint || checkpoint.startsWith("external::"); + const usableCheckpoint = + !!checkpoint && !checkpoint.startsWith("external::") && !looksLikePath(checkpoint); + const needsCatalog = !usableCheckpoint; // Only when the checkpoint can't answer it: /v1/models scans the model dirs and // HF caches. Re-runs when a model is loaded or unloaded. @@ -378,18 +394,23 @@ function useExampleModelName(): string { }, [needsCatalog]); return useMemo(() => { - if (checkpoint && !checkpoint.startsWith("external::")) { + if (usableCheckpoint && checkpoint) { if (ggufVariant && !checkpoint.includes(":")) { return `${checkpoint}:${ggufVariant}`; } return checkpoint; } - // No local checkpoint (or an external provider, which /v1 cannot serve): - // name something this server actually holds. - return ( - catalog.find((m) => m.loaded)?.id ?? catalog[0]?.id ?? MODEL_FALLBACK - ); - }, [catalog, checkpoint, ggufVariant]); + // No usable checkpoint (none, an external provider, or a raw load path): + // name something this server actually holds, quant included so the request + // pins the file on disk instead of letting the server pick a quant. + const pick = catalog.find((m) => m.loaded) ?? catalog[0]; + if (!pick) { + return MODEL_FALLBACK; + } + return pick.quant && !pick.id.includes(":") + ? `${pick.id}:${pick.quant}` + : pick.id; + }, [catalog, checkpoint, ggufVariant, usableCheckpoint]); } // Backend PATH detection is only safe in the desktop app, where the UI owns diff --git a/studio/frontend/src/features/settings/tabs/api-keys-tab.tsx b/studio/frontend/src/features/settings/tabs/api-keys-tab.tsx index f1e503f7a3f..7a1fdc1fb23 100644 --- a/studio/frontend/src/features/settings/tabs/api-keys-tab.tsx +++ b/studio/frontend/src/features/settings/tabs/api-keys-tab.tsx @@ -168,12 +168,12 @@ export function ApiKeysTab() { )} + + - - !o && setRevokeTarget(null)}> diff --git a/studio/frontend/src/i18n/locales/en.ts b/studio/frontend/src/i18n/locales/en.ts index 2e6571c3002..1b7a11213fa 100644 --- a/studio/frontend/src/i18n/locales/en.ts +++ b/studio/frontend/src/i18n/locales/en.ts @@ -282,20 +282,18 @@ export const en = { sectionTitle: "Model auto-switch (OpenAI API)", enable: "Switch model by request", enableDescription: - "When an OpenAI-compatible request names a different downloaded GGUF, load it before serving. Off by default; unknown names keep serving the loaded model.", + "Load a downloaded GGUF named in an API request before serving. Off by default.", idleUnload: "Idle auto-unload", idleUnloadDescription: - "Unload the model after this many idle seconds to free VRAM; the next request reloads it. 0 keeps it loaded. Minimum 60 seconds.", - idleNeedsEnable: - "Turn on Switch model by request so an unloaded model reloads on next use.", - idleActiveViaEnv: - "Idle auto-unload is active via the UNSLOTH_MODEL_IDLE_TTL environment variable.", + "Free VRAM after this many idle seconds. 0 keeps it loaded, minimum 60.", + idleNeedsEnable: "Turn on Switch model by request first.", + idleActiveViaEnv: "Active via UNSLOTH_MODEL_IDLE_TTL.", loadError: "Failed to load model auto-switch settings.", saveError: "Failed to save model auto-switch settings.", idleError: "Enter 0 to keep the model loaded, or at least 60 seconds.", keepKv: "Keep chat context across idle unload", keepKvDescription: - "Save the model's KV cache to disk before an idle unload and restore it on reload, so resumed chats skip re-reading their history. Chat context is written to disk (up to 10 GB) until it is restored or cleaned up.", + "Save the KV cache before an idle unload so resumed chats skip re-reading history. Up to 10 GB on disk.", }, previewSharing: { sectionTitle: "Preview sharing", diff --git a/tests/studio/test_usage_examples_model_source_contract.py b/tests/studio/test_usage_examples_model_source_contract.py index 549957ea60c..8e355de3a19 100644 --- a/tests/studio/test_usage_examples_model_source_contract.py +++ b/tests/studio/test_usage_examples_model_source_contract.py @@ -25,7 +25,9 @@ def test_examples_name_a_model_the_server_can_serve(): assert "listOpenAIModels()" in hook # Precedence: live checkpoint, then a loaded catalog entry, then any entry, # and only then the placeholder. - assert "catalog.find((m) => m.loaded)?.id ?? catalog[0]?.id ?? MODEL_FALLBACK" in hook + assert "catalog.find((m) => m.loaded) ?? catalog[0]" in hook + # The snippet pins the quant so the request names the file on disk. + assert "`${pick.id}:${pick.quant}`" in hook api = OPENAI_MODELS_TS.read_text(encoding = "utf-8") assert 'authFetch("/v1/models")' in api @@ -45,3 +47,34 @@ def test_usage_examples_has_no_duplicate_auto_switch_control(): tab = API_KEYS_TAB_TSX.read_text(encoding = "utf-8") assert "" in tab + + +API_MONITOR_TSX = SETTINGS / "components/api-monitor-console.tsx" + + +def test_api_monitor_pages_five_at_a_time(): + # 50 terminal entries are retained backend-side; the console used to dump all + # of them into one scroller. + src = API_MONITOR_TSX.read_text(encoding = "utf-8") + assert "const PAGE_SIZE = 5;" in src + assert "ordered.slice(" in src + # Paging back must freeze the id order, or live traffic reorders history + # under the cursor between polls. + assert "frozenIds" in src + assert "setFrozenIds((prev) => prev ?? entries.map((entry) => entry.id))" in src + + +def test_api_monitor_renders_lifecycle_rows(): + src = API_MONITOR_TSX.read_text(encoding = "utf-8") + assert "function LifecycleEntry(" in src + assert 'entry.kind === "lifecycle"' in src + for label in ("Loading model", "Model loaded", "Model unloaded"): + assert label in src + # Lifecycle rows have no prompt/reply to fetch. + assert "isLifecycle(entry) || !expandedIds.has(entry.id)" in src + + +def test_auto_switch_section_sits_above_the_monitor(): + tab = API_KEYS_TAB_TSX.read_text(encoding = "utf-8") + assert tab.index("") < tab.index("") + assert tab.index("") < tab.index(" Date: Sun, 26 Jul 2026 07:33:28 +0000 Subject: [PATCH 04/56] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/tests/test_api_monitor.py | 4 +--- studio/backend/tests/test_openai_catalog.py | 3 +-- 2 files changed, 2 insertions(+), 5 deletions(-) diff --git a/studio/backend/tests/test_api_monitor.py b/studio/backend/tests/test_api_monitor.py index 44b1f5fecde..6a7b228f59d 100644 --- a/studio/backend/tests/test_api_monitor.py +++ b/studio/backend/tests/test_api_monitor.py @@ -355,7 +355,5 @@ def test_lifecycle_rows_share_the_retention_budget(): def test_request_rows_report_kind_request(): monitor = ApiMonitor(max_entries = 2) - monitor.start( - endpoint = "/v1/chat/completions", method = "POST", model = "m", prompt = "hi" - ) + monitor.start(endpoint = "/v1/chat/completions", method = "POST", model = "m", prompt = "hi") assert monitor.snapshot()[0]["kind"] == "request" diff --git a/studio/backend/tests/test_openai_catalog.py b/studio/backend/tests/test_openai_catalog.py index 3c5c953b962..9f44eddd281 100644 --- a/studio/backend/tests/test_openai_catalog.py +++ b/studio/backend/tests/test_openai_catalog.py @@ -242,8 +242,7 @@ def test_lifecycle_label_recovers_the_repo_id_from_an_hf_cache_path(): # sha, so the row would otherwise be labelled with a hash. snap = "/home/me/.cache/huggingface/hub/models--unsloth--gemma-4-E4B-it-GGUF/snapshots/bfc15c3" assert ( - inf._lifecycle_model_label(snap, "UD-Q4_K_XL") - == "unsloth/gemma-4-E4B-it-GGUF:UD-Q4_K_XL" + inf._lifecycle_model_label(snap, "UD-Q4_K_XL") == "unsloth/gemma-4-E4B-it-GGUF:UD-Q4_K_XL" ) From d80cb3a121818c955816512edfce54eb4580ce7d Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 26 Jul 2026 09:51:53 +0000 Subject: [PATCH 05/56] Studio: optionally download a model named in an OpenAI API request Auto-switch only ever loaded models already on disk, so naming one this server does not have either 404s or, when something else is loaded, gets quietly answered by the resident model. Add openai_api_auto_download_model (off by default, gated on auto-switch). When on, a /v1 request naming a GGUF repo that is not downloaded starts a background fetch and returns 503 with Retry-After and a typed model_downloading code. The resident model keeps serving in the meantime, and the retry after the download completes is served by the new model through the existing auto-switch path. The download reuses the Hub manager's service layer, which already does repo-id validation, casing, claim bookkeeping, disk preflight, resume and cancel. The in-loader download is deliberately not used: it silently falls back to a smaller quant under low disk, which is wrong when the caller named an exact one. Admission is narrow, since a request only needs an API key: - namespace/name only, so gpt-4 and other foreign ids fall through to the resident model exactly as before - GGUF only, decided from the remote file list rather than the repo name - anything declaring auto_map is refused, so trust_remote_code stays a deliberate opt-in in the UI and can never be granted over the API - a single download at a time, plus a free-disk reserve - one model_info call answers existence, gating and the quant list, so a missing repo, a gated repo and a wrong quant each get their own error With the setting off every one of these paths is byte-identical to before. Also: - monitor rows for downloads, with a live percentage - public_model_id resolves an HF cache snapshot to its repo id, so a cache-loaded model is no longer labelled with a commit sha; this drops the duplicate helper added for the monitor and fixes the same leak in the inference status response - the unedited sk-unsloth-YOUR_KEY from the copyable examples now says so instead of "Invalid or expired API key"; every other bad key keeps the generic message --- studio/backend/auth/authentication.py | 19 +- studio/backend/core/inference/api_monitor.py | 13 + .../core/inference/local_model_resolver.py | 8 + studio/backend/core/inference/model_ids.py | 23 +- .../core/inference/openai_auto_download.py | 488 ++++++++++++++++++ studio/backend/routes/inference.py | 91 +++- studio/backend/routes/settings.py | 15 +- studio/backend/tests/test_model_ids.py | 18 + .../tests/test_openai_auto_download.py | 470 +++++++++++++++++ .../backend/tests/test_openai_auto_switch.py | 15 +- .../utils/openai_auto_switch_settings.py | 44 +- .../frontend/src/features/chat/types/api.ts | 10 +- .../settings/api/openai-auto-switch.ts | 11 + .../components/api-monitor-console.tsx | 13 +- .../components/model-auto-switch-section.tsx | 19 + studio/frontend/src/i18n/locales/en.ts | 3 + ...st_usage_examples_model_source_contract.py | 29 ++ 17 files changed, 1256 insertions(+), 33 deletions(-) create mode 100644 studio/backend/core/inference/openai_auto_download.py create mode 100644 studio/backend/tests/test_openai_auto_download.py diff --git a/studio/backend/auth/authentication.py b/studio/backend/auth/authentication.py index dfb8fc513e3..cdb2b897c50 100644 --- a/studio/backend/auth/authentication.py +++ b/studio/backend/auth/authentication.py @@ -164,6 +164,23 @@ async def get_current_subject_allow_password_change( ) +# The literal the copyable API examples ship with. Pasting a snippet unedited is +# a much likelier mistake than a revoked key, so say so instead of "invalid". +API_KEY_PLACEHOLDER = f"{API_KEY_PREFIX}YOUR_KEY" + + +def _invalid_api_key_detail(token: str) -> str: + """Why the key failed. Only the unedited example placeholder is called out; + every real key still gets one indistinguishable message, so this reveals + nothing about which keys exist.""" + if token == API_KEY_PLACEHOLDER: + return ( + "This is the placeholder key from the example. Create an API key in " + f"Unsloth Studio under Settings > API and use it in place of {API_KEY_PLACEHOLDER}." + ) + return "Invalid or expired API key" + + async def _get_current_subject( credentials: HTTPAuthorizationCredentials, *, allow_password_change: bool ) -> str: @@ -176,7 +193,7 @@ async def _get_current_subject( if username is None: raise HTTPException( status_code = status.HTTP_401_UNAUTHORIZED, - detail = "Invalid or expired API key", + detail = _invalid_api_key_detail(token), ) return username diff --git a/studio/backend/core/inference/api_monitor.py b/studio/backend/core/inference/api_monitor.py index 2390d235a2e..e2a72f31419 100644 --- a/studio/backend/core/inference/api_monitor.py +++ b/studio/backend/core/inference/api_monitor.py @@ -59,6 +59,8 @@ class ApiMonitorEntry: event: Optional[str] = None reason: Optional[str] = None shared: bool = False + # 0-100 for a running download row; None when not applicable. + progress: Optional[float] = None def snapshot(self, *, include_details: bool = True) -> dict[str, Any]: duration_ms = None @@ -95,6 +97,7 @@ def snapshot(self, *, include_details: bool = True) -> dict[str, Any]: "kind": self.kind, "event": self.event, "reason": self.reason, + "progress": self.progress, } if include_details: payload["prompt"] = self.prompt @@ -186,6 +189,16 @@ def relabel(self, entry_id: Optional[str], model: str) -> None: entry.model = model entry.updated_at = time.time() + def set_progress(self, entry_id: Optional[str], progress: Optional[float]) -> None: + """Update an open download row's percentage (clamped to 0-100).""" + if not entry_id or progress is None: + return + with self._lock: + entry = self._find_locked(entry_id) + if entry is not None and entry.status == "running": + entry.progress = min(100.0, max(0.0, float(progress))) + entry.updated_at = time.time() + def discard(self, entry_id: Optional[str]) -> None: """Drop a row that turned out not to be an event (a load that was already satisfied, so nothing was actually loaded).""" diff --git a/studio/backend/core/inference/local_model_resolver.py b/studio/backend/core/inference/local_model_resolver.py index 0533db944f6..61e4db8f4ea 100644 --- a/studio/backend/core/inference/local_model_resolver.py +++ b/studio/backend/core/inference/local_model_resolver.py @@ -232,6 +232,14 @@ def _scan_hf_once(directory) -> list: return index +def invalidate_index() -> None: + """Drop the cached scan so the next resolve sees a just-finished download, + rather than waiting out the TTL.""" + global _scan + with _lock: + _scan = (0.0, {}) + + def _index() -> dict[str, _LocalGgufEntry]: global _scan # Build under the lock so concurrent callers with an expired cache don't all diff --git a/studio/backend/core/inference/model_ids.py b/studio/backend/core/inference/model_ids.py index 548cc60f94a..07148402dd7 100644 --- a/studio/backend/core/inference/model_ids.py +++ b/studio/backend/core/inference/model_ids.py @@ -39,10 +39,28 @@ def _looks_like_path(identifier: str) -> bool: return False +def hf_cache_repo_id(path: Optional[str]) -> Optional[str]: + """``.../models--org--name/snapshots/`` -> ``org/name``, else None. + + A model loaded straight out of the HF cache has a snapshot directory as its + identifier, whose basename is a commit hash. Recover the repo id so callers + show ``unsloth/gemma-4-31B-it-GGUF`` rather than ``c1ac76e99d55...``. + """ + if not path: + return None + for part in str(path).replace("\\", "/").split("/"): + if part.startswith("models--"): + return part[len("models--") :].replace("--", "/") + return None + + def public_model_id(identifier: Optional[str]) -> Optional[str]: """Return a clean, path-free public id for *identifier*. - - Local GGUF path -> the file stem with ``.gguf`` stripped, e.g. + - HF cache path -> the repo id it came from, e.g. + ``~/.cache/huggingface/hub/models--unsloth--X-GGUF/snapshots/`` -> + ``unsloth/X-GGUF``. + - Other local GGUF path -> the file stem with ``.gguf`` stripped, e.g. ``/srv/models/Qwen3-30B-A3B-Q4_K_M.gguf`` -> ``Qwen3-30B-A3B-Q4_K_M``. - HF repo id (``org/model``) and already-clean names -> returned unchanged. - ``None`` / empty -> returned unchanged. @@ -51,6 +69,9 @@ def public_model_id(identifier: Optional[str]) -> Optional[str]: return identifier if not _looks_like_path(identifier): return identifier + repo_id = hf_cache_repo_id(identifier) + if repo_id: + return repo_id name = os.path.basename(identifier.replace("\\", "/").rstrip("/")) if name.lower().endswith(_GGUF_SUFFIX): name = name[: -len(_GGUF_SUFFIX)] diff --git a/studio/backend/core/inference/openai_auto_download.py b/studio/backend/core/inference/openai_auto_download.py new file mode 100644 index 00000000000..3f6670c14e8 --- /dev/null +++ b/studio/backend/core/inference/openai_auto_download.py @@ -0,0 +1,488 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0 + +"""Opt-in: fetch a GGUF a /v1 request names but this server doesn't have. + +Auto-switch only loads models already on disk. With +``openai_api_auto_download_model`` on, a miss that looks like a real Hub repo is +downloaded in the background instead of erroring, and the request is told to +retry rather than being held open: a quant is routinely tens of GB, far longer +than any client (or the Cloudflare edge on ``--secure``) will wait, and the +inference lifecycle gate must not be held meanwhile. The resident model keeps +serving throughout, and the retry that lands after the download is served by the +new model through the ordinary auto-switch path. + +Admission is deliberately narrow, since a request only needs an API key: +- ``namespace/name`` only. A bare name like ``gpt-4`` falls through to the + resident model exactly as before, so drop-in clients are unaffected. +- GGUF repos only, decided from the remote file list, not the repo name. GGUF + runs under llama.cpp, which never imports repo Python. +- Anything declaring ``auto_map`` is refused, so ``trust_remote_code`` can only + ever be granted deliberately in the UI, never by an API call. +- One download at a time, so a key holder cannot fan out fetches. +""" + +from __future__ import annotations + +import asyncio +import shutil +import threading +import time +from dataclasses import dataclass +from typing import Optional + +from loggers import get_logger + +logger = get_logger(__name__) + +# Hub metadata is one small request; keep it short so a slow Hub can't stall the +# request path for long. +_MODEL_INFO_TIMEOUT_S = 8.0 +# Headroom left free after the download, so filling the disk can't wedge the box. +_DISK_RESERVE_BYTES = 5 * 1024**3 +_WATCH_POLL_S = 2.0 +# A stalled watcher must not pin the single-flight slot forever. +_MAX_WATCH_S = 24 * 60 * 60 +_RETRY_AFTER_S = 30 +_MAX_LISTED_VARIANTS = 8 + + +@dataclass(frozen = True) +class AutoDownloadRefusal: + """Why this request cannot be served yet. The route turns it into an + HTTPException with the surface's own error envelope.""" + + status: int + code: str + message: str + retry_after: Optional[int] = None + + +@dataclass +class _Active: + repo_id: str + # None while the Hub probe is still deciding which quant to fetch. + variant: Optional[str] = None + expected_bytes: int = 0 + monitor_id: Optional[str] = None + started_at: float = 0.0 + + +_lock = threading.Lock() +_active: Optional[_Active] = None + + +def _public_label(repo_id: str, variant: Optional[str]) -> str: + return f"{repo_id}:{variant}" if variant else repo_id + + +def split_model_ref(requested: str) -> tuple[str, Optional[str]]: + """``org/repo:QUANT`` -> ``("org/repo", "QUANT")``; no suffix -> variant None. + + Splits on the last colon only when the suffix carries no slash, so a Windows + drive letter or a path never reads as a quant. + """ + text = (requested or "").strip() + base, sep, suffix = text.rpartition(":") + if not sep or not base or "/" in suffix or not suffix: + return text, None + return base.strip(), suffix.strip() + + +def is_downloadable_ref(requested: str) -> bool: + """Whether *requested* is shaped like a Hub repo we may fetch. + + Requires an explicit namespace. That keeps ``gpt-4`` and other foreign ids + falling through untouched, and avoids the bare-name ``unsloth/`` prefixing in + ModelConfig.from_identifier turning an unrelated label into a real repo. + """ + from hub.utils.paths import is_valid_repo_id + + repo_id, variant = split_model_ref(requested) + if "/" not in repo_id or not is_valid_repo_id(repo_id): + return False + if variant is not None: + from hub.utils.paths import is_valid_gguf_variant + return is_valid_gguf_variant(variant) + return True + + +def _gguf_variants(siblings) -> dict[str, int]: + """Quant label -> total bytes, from a model_info sibling list. + + Mirrors list_gguf_variants: companions (mmproj/MTP) and big-endian builds + are not selectable quants, and sharded quants sum across their shards. + """ + from utils.models.model_config import ( + _extract_quant_label, + _is_big_endian_gguf_path, + _is_mmproj, + _is_mtp_drafter, + ) + + sizes: dict[str, int] = {} + for sibling in siblings or []: + name = getattr(sibling, "rfilename", "") or "" + if not name.lower().endswith(".gguf"): + continue + quant = _extract_quant_label(name) + if _is_mmproj(name) or _is_mtp_drafter(name) or _is_big_endian_gguf_path(name, quant): + continue + sizes[quant] = sizes.get(quant, 0) + int(getattr(sibling, "size", 0) or 0) + return sizes + + +def _enough_disk(need_bytes: int) -> tuple[bool, int]: + """(fits, free_bytes). Fail-open on an unreadable cache root: the download + worker runs its own preflight, this only adds the reserve margin.""" + try: + from hub.utils.hf_cache_state import hf_cache_root + + root = hf_cache_root(create = True) + if root is None: + return True, 0 + free = shutil.disk_usage(root).free + except Exception: + return True, 0 + return free >= need_bytes + _DISK_RESERVE_BYTES, free + + +def _gb(num_bytes: int) -> str: + return f"{num_bytes / 1024**3:.1f} GB" + + +async def _job_state(repo_id: str, variant: Optional[str]) -> tuple[str, Optional[str]]: + from hub.services.models import downloads + try: + status = await downloads.get_download_status_response(repo_id, variant or "") + return status.state, status.error + except Exception as exc: + logger.debug("auto-download: status probe failed for %r: %s", repo_id, exc) + return "idle", None + + +async def _progress_percent( + repo_id: str, variant: Optional[str], expected_bytes: int, hf_token: Optional[str] +) -> Optional[float]: + """0-100, or None. The hub service reports a 0-1 fraction, so scale it.""" + from hub.services.models import downloads + try: + payload = await downloads.get_gguf_download_progress_response( + repo_id, variant or "", expected_bytes, hf_token + ) + fraction = payload.get("progress") + if not isinstance(fraction, (int, float)): + return None + return min(100.0, max(0.0, float(fraction) * 100.0)) + except Exception: + return None + + +def _release(repo_id: str) -> None: + global _active + with _lock: + if _active is not None and _active.repo_id == repo_id: + _active = None + + +async def _watch(active: _Active, hf_token: Optional[str]) -> None: + """Poll a dispatched job so the monitor row resolves and the resolver cache + is dropped the moment the weights land.""" + from core.inference import api_monitor as monitor_module + from core.inference.local_model_resolver import invalidate_index + + api_monitor = monitor_module.api_monitor + deadline = time.monotonic() + _MAX_WATCH_S + try: + while time.monotonic() < deadline: + await asyncio.sleep(_WATCH_POLL_S) + state, error = await _job_state(active.repo_id, active.variant) + if state == "running": + api_monitor.set_progress( + active.monitor_id, + await _progress_percent( + active.repo_id, active.variant, active.expected_bytes, hf_token + ), + ) + continue + if state == "complete": + # Drop the 5s resolver cache so the next retry resolves the new + # model instead of missing it again. + await asyncio.to_thread(invalidate_index) + api_monitor.finish(active.monitor_id, status = "completed") + elif state == "idle": + # The job vanished without a terminal state (worker killed). + api_monitor.fail_open(active.monitor_id, "Download did not complete") + else: + api_monitor.fail_open(active.monitor_id, error or f"Download {state}") + return + api_monitor.fail_open(active.monitor_id, "Download timed out") + except asyncio.CancelledError: + raise + except Exception as exc: + logger.warning("auto-download: watcher failed for %r: %s", active.repo_id, exc) + api_monitor.fail_open(active.monitor_id, "Download tracking failed") + finally: + _release(active.repo_id) + + +def _downloading_refusal(label: str, percent: Optional[float]) -> AutoDownloadRefusal: + progress = f" ({percent:.0f}% done)" if percent is not None else "" + return AutoDownloadRefusal( + status = 503, + code = "model_downloading", + message = (f"Downloading '{label}'{progress}. Retry shortly. Track it in Unsloth Studio."), + retry_after = _RETRY_AFTER_S, + ) + + +async def maybe_auto_download( + requested_model: str, *, hf_token: Optional[str] = None +) -> Optional[AutoDownloadRefusal]: + """Start (or report on) a background fetch of *requested_model*. + + Returns None when the request should carry on unchanged, or a refusal the + caller must raise. Only called after the local resolver has already missed. + """ + global _active + + repo_id, wanted_variant = split_model_ref(requested_model) + if not is_downloadable_ref(requested_model): + return None + + # Adopt or reject against the single in-flight slot before touching the + # network, so retries during a long download stay free. + with _lock: + current = _active + if current is not None and current.repo_id == repo_id: + adopted = current + elif current is not None: + return AutoDownloadRefusal( + status = 503, + code = "model_download_busy", + message = ( + f"Already downloading '{_public_label(current.repo_id, current.variant)}'. " + f"Retry '{requested_model}' once it finishes." + ), + retry_after = _RETRY_AFTER_S, + ) + else: + adopted = None + _active = _Active(repo_id = repo_id, started_at = time.time()) + + if adopted is not None: + state, error = await _job_state(adopted.repo_id, adopted.variant) + if state in ("running", "cancelling"): + return _downloading_refusal( + _public_label(adopted.repo_id, adopted.variant), + await _progress_percent( + adopted.repo_id, adopted.variant, adopted.expected_bytes, hf_token + ), + ) + if state == "error": + # Surface once, then free the slot so a retry can start over. + _release(adopted.repo_id) + return AutoDownloadRefusal( + status = 502, + code = "model_download_failed", + message = f"Downloading '{requested_model}' failed: {error or 'unknown error'}", + ) + if adopted.variant is None: + # Another request is still probing this same repo. + return _downloading_refusal(adopted.repo_id, None) + # complete/idle: the watcher is about to release; ask for one more retry. + return _downloading_refusal(_public_label(adopted.repo_id, adopted.variant), 100.0) + + try: + return await _admit_and_start(repo_id, wanted_variant, requested_model, hf_token) + except Exception: + _release(repo_id) + raise + + +async def _admit_and_start( + repo_id: str, wanted_variant: Optional[str], requested_model: str, hf_token: Optional[str] +) -> Optional[AutoDownloadRefusal]: + from hub.utils.hf_errors import hf_error_status + + def _probe(): + from huggingface_hub import HfApi + return HfApi(token = hf_token).model_info( + repo_id, files_metadata = True, timeout = _MODEL_INFO_TIMEOUT_S + ) + + try: + info = await asyncio.to_thread(_probe) + except Exception as exc: + _release(repo_id) + status = hf_error_status(exc) + if status == 403: + return AutoDownloadRefusal( + status = 403, + code = "model_access_denied", + message = ( + f"'{repo_id}' is gated on Hugging Face. Accept its licence and add an " + "access token in Unsloth Studio, then retry." + ), + ) + if status == 404: + # A private repo reads as absent without a token; don't confirm either way. + return AutoDownloadRefusal( + status = 404, + code = "model_not_found", + message = f"'{repo_id}' was not found on Hugging Face, or is not accessible.", + ) + logger.warning("auto-download: Hub lookup failed for %r: %s", repo_id, exc) + return AutoDownloadRefusal( + status = 503, + code = "model_lookup_failed", + message = f"Could not reach Hugging Face to look up '{repo_id}'. Retry shortly.", + retry_after = _RETRY_AFTER_S, + ) + + variants = _gguf_variants(getattr(info, "siblings", None)) + if not variants: + _release(repo_id) + return AutoDownloadRefusal( + status = 400, + code = "model_not_supported", + message = ( + f"'{repo_id}' has no GGUF weights. Automatic download serves GGUF only; " + "load other formats from Unsloth Studio." + ), + ) + + # trust_remote_code gate. _config_has_auto_map is tri-state: refuse on True + # and on None (unreadable), rather than _requires_trust_remote_code_for_model, + # which swallows errors into False. Fine as a UI hint, wrong as an admission. + from utils.security.consent import _config_has_auto_map + + has_auto_map = await asyncio.to_thread(_config_has_auto_map, repo_id, hf_token) + if has_auto_map is not False: + _release(repo_id) + unknown = has_auto_map is None + return AutoDownloadRefusal( + status = 403, + code = "remote_code_consent_required", + message = ( + f"'{repo_id}' " + + ( + "could not be checked for custom code" + if unknown + else "ships custom code that runs on load" + ) + + ". Load it once in Unsloth Studio to review and approve it, then retry." + ), + ) + + variant = _match_variant(wanted_variant, variants) + if variant is None: + _release(repo_id) + listed = sorted(variants) + shown = ", ".join(listed[:_MAX_LISTED_VARIANTS]) + extra = len(listed) - _MAX_LISTED_VARIANTS + return AutoDownloadRefusal( + status = 404, + code = "model_not_found", + message = ( + f"'{repo_id}' has no quant '{wanted_variant}'. Available quants: " + f"{shown}{f' and {extra} more' if extra > 0 else ''}." + ), + ) + + expected_bytes = variants[variant] + fits, free = _enough_disk(expected_bytes) + if not fits: + _release(repo_id) + return AutoDownloadRefusal( + status = 507, + code = "insufficient_disk_space", + message = ( + f"'{_public_label(repo_id, variant)}' needs {_gb(expected_bytes)} plus " + f"{_gb(_DISK_RESERVE_BYTES)} headroom, but only {_gb(free)} is free." + ), + ) + + return await _dispatch(repo_id, variant, expected_bytes, requested_model, hf_token) + + +def _match_variant(wanted: Optional[str], variants: dict[str, int]) -> Optional[str]: + """Resolve the requested quant against what the repo actually has. + + An explicit quant matches case-insensitively and must exist: never quietly + substitute another, unlike the loader's low-disk fallback. A bare repo id + uses the same preference order as a manual load. + """ + if wanted: + lowered = {name.lower(): name for name in variants} + return lowered.get(wanted.strip().lower()) + from utils.models.model_config import _pick_best_gguf + + # _pick_best_gguf ranks filenames by quant substring, so give it synthesized + # "
{formatTime(entry.started_at)}
- {entry.event === "load" ? ( + {entry.event === "load" || entry.event === "download" ? (
{formatDuration(entry.duration_ms)}
) : null}
diff --git a/studio/frontend/src/features/settings/components/model-auto-switch-section.tsx b/studio/frontend/src/features/settings/components/model-auto-switch-section.tsx index aa6857cff53..6ebd12ce3c0 100644 --- a/studio/frontend/src/features/settings/components/model-auto-switch-section.tsx +++ b/studio/frontend/src/features/settings/components/model-auto-switch-section.tsx @@ -65,6 +65,7 @@ export function ModelAutoSwitchSection() { idleSeconds: number | undefined, syncDraft = true, keepKv?: boolean, + autoDownload?: boolean, ) => { setIsSaving(true); setError(null); @@ -73,6 +74,7 @@ export function ModelAutoSwitchSection() { enabled, idleSeconds, keepKv, + autoDownload, ); setSettings(saved); if (syncDraft) { @@ -117,6 +119,11 @@ export function ModelAutoSwitchSection() { void persist(settings.enabled, undefined, false, keepKv); }; + const handleAutoDownloadToggle = (autoDownload: boolean) => { + if (!settings) return; + void persist(settings.enabled, undefined, false, undefined, autoDownload); + }; + return ( + + + ") < tab.index("") assert tab.index("") < tab.index("")] + + +def test_auto_download_copy_warns_about_api_key_holders(): + en = EN_TS.read_text(encoding = "utf-8") + start = en.find("autoDownloadDescription:") + assert start != -1 + description = en[start : en.find("\n", en.find('",', start))] + assert "API key" in description From d61ca7b08f0fefd3ce75776cf2b2d6b5ee3689d2 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Sun, 26 Jul 2026 09:54:45 +0000 Subject: [PATCH 06/56] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/tests/test_openai_auto_switch.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/studio/backend/tests/test_openai_auto_switch.py b/studio/backend/tests/test_openai_auto_switch.py index 36c369a314f..0049f0f97c7 100644 --- a/studio/backend/tests/test_openai_auto_switch.py +++ b/studio/backend/tests/test_openai_auto_switch.py @@ -289,9 +289,9 @@ def test_openai_compat_routes_bound_to_handlers_with_auth(): for key, handler in expected.items(): assert key in seen, f"route {key} is not registered" route = seen[key] - assert route.endpoint.__name__ == handler, ( - f"{key} bound to {route.endpoint.__name__}, expected {handler}" - ) + assert ( + route.endpoint.__name__ == handler + ), f"{key} bound to {route.endpoint.__name__}, expected {handler}" deps = [d.call.__name__ for d in route.dependant.dependencies] assert "get_current_subject" in deps, f"{key} lost its auth dependency" From 658f106fc2f8d18a3172fb3d20db14ece3dbd6ed Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 26 Jul 2026 10:18:28 +0000 Subject: [PATCH 07/56] Studio: add an Unload button to the API monitor The monitor names the loaded model but offered no way to free it. Idle auto-unload is the only existing release path, and it needs a TTL and a wait. The button sits next to Refresh, appears only while a model is loaded and is disabled mid-unload. /unload matches on the internal identifier, which this response deliberately omits because it would be a host path, so the click reads it from /api/inference/status the same way the chat runtime does rather than widening the monitor payload. Also stamp the manual unload row with the quant, read before the teardown clears it, so it reads repo:QUANT like the load row it pairs with. --- studio/backend/routes/inference.py | 6 ++- .../components/api-monitor-console.tsx | 46 ++++++++++++++++++- ...st_usage_examples_model_source_contract.py | 12 +++++ 3 files changed, 62 insertions(+), 2 deletions(-) diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 25a157308f4..bfdd3a1e412 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -5831,13 +5831,17 @@ async def unload_model(request: UnloadRequest, current_subject: str = Depends(ge ) or not llama_backend.is_loaded ): + # Read the identity before the teardown clears it, so the row reads + # repo:QUANT like the matching load row rather than a bare path. + _unloaded = _llama_public_model_id(llama_backend, request.model_path) + _unloaded_variant = getattr(llama_backend, "hf_variant", None) # A manual unload is a deliberate user action: tear down now even if a # request is mid-stream (only the automatic idle loop defers to it). llama_backend.unload_model() note_model_unloaded() api_monitor.record_lifecycle( event = "unload", - model = _lifecycle_model_label(request.model_path), + model = _lifecycle_model_label(_unloaded, _unloaded_variant), reason = "manual", ) logger.info(f"Unloaded GGUF model: {request.model_path}") diff --git a/studio/frontend/src/features/settings/components/api-monitor-console.tsx b/studio/frontend/src/features/settings/components/api-monitor-console.tsx index 03ab95bdd62..0d242e59503 100644 --- a/studio/frontend/src/features/settings/components/api-monitor-console.tsx +++ b/studio/frontend/src/features/settings/components/api-monitor-console.tsx @@ -7,6 +7,7 @@ import { ActivityIcon, ChevronDownIcon, CircleIcon, + PowerOffIcon, RefreshCwIcon, } from "lucide-react"; import { @@ -17,7 +18,13 @@ import { useRef, useState, } from "react"; -import { getApiMonitor, getApiMonitorEntry } from "../../chat/api/chat-api"; +import { + getApiMonitor, + getApiMonitorEntry, + getInferenceStatus, + unloadModel, +} from "../../chat/api/chat-api"; +import { resolveInferenceCheckpointId } from "../../chat/lib/apply-inference-status-to-store"; import type { ApiMonitorEntry, ApiMonitorResponse } from "../../chat/types/api"; const API_INFERENCE_PREFIX_RE = /^\/api\/inference/; @@ -250,6 +257,7 @@ export function ApiMonitorConsole(): ReactElement { const [data, setData] = useState(null); const [error, setError] = useState(null); const [refreshing, setRefreshing] = useState(false); + const [unloading, setUnloading] = useState(false); const [expandedIds, setExpandedIds] = useState>(() => new Set()); const [details, setDetails] = useState>({}); const [loadingDetails, setLoadingDetails] = useState>( @@ -270,6 +278,29 @@ export function ApiMonitorConsole(): ReactElement { } }, []); + // Free the VRAM the loaded model is holding. The monitor only knows the model's + // public id, and /unload matches on the internal one, so read that from status + // the same way the chat runtime does rather than putting a host path in this + // response. With auto-switch on the next API request loads a model again. + const unloadActiveModel = useCallback(async (): Promise => { + setUnloading(true); + try { + const status = await getInferenceStatus(); + const checkpoint = resolveInferenceCheckpointId(status); + if (!checkpoint) { + setError(null); + return; + } + await unloadModel({ model_path: checkpoint }); + setError(null); + await loadMonitor(); + } catch (err: unknown) { + setError(err instanceof Error ? err.message : "Failed to unload the model"); + } finally { + setUnloading(false); + } + }, [loadMonitor]); + useEffect(() => { let cancelled = false; let timer: number | undefined; @@ -445,6 +476,19 @@ export function ApiMonitorConsole(): ReactElement {
{statusLabel}
+ {data?.active_model ? ( + + ) : null} - ) : null} + {/* Always rendered, disabled when idle: hiding it made the only manual + release path invisible exactly when someone goes looking for it. */} + diff --git a/studio/frontend/src/features/settings/components/usage-examples.tsx b/studio/frontend/src/features/settings/components/usage-examples.tsx index 6bd74f1371b..5acdb6911ae 100644 --- a/studio/frontend/src/features/settings/components/usage-examples.tsx +++ b/studio/frontend/src/features/settings/components/usage-examples.tsx @@ -329,8 +329,10 @@ function buildSnippets( } const KEY_PLACEHOLDER = "sk-unsloth-YOUR_KEY"; -const MODEL_FALLBACK = "unsloth/gemma-4-E4B-it-GGUF:UD-Q5_K_XL"; const USE_TUNNEL_KEY = "unsloth_api_use_tunnel"; +// Slow retry while /v1 has nothing to name: a download or a load can finish +// while this panel is open, and neither moves the checkpoint below. +const CATALOG_RETRY_MS = 15000; function readUseTunnelPref(): boolean { if (typeof window === "undefined") return true; @@ -364,34 +366,48 @@ function looksLikePath(id: string): boolean { ); } -// The model the examples name. With nothing loaded this used to print -// MODEL_FALLBACK, a repo id the user likely never downloaded, so the copied -// snippet 404d. Prefer a real id from the catalog /v1 resolves against; the -// constant is only the floor for a server with no models, so it is never blank. -function useExampleModelName(): string { +// The model the examples name. With nothing loaded this used to print a +// hardcoded repo id the user likely never downloaded, so the copied snippet +// 404d. Only ever name an id /v1 resolves against; null means this server has +// nothing to serve yet and the panel says so instead of printing a dead id. +function useExampleModelName(): string | null { const checkpoint = useChatRuntimeStore((s) => s.params.checkpoint); const ggufVariant = useChatRuntimeStore((s) => s.activeGgufVariant); - const [catalog, setCatalog] = useState([]); + // null until /v1/models answers: "not asked yet" must not read as "this + // server holds nothing", or the first render already picks a name. + const [catalog, setCatalog] = useState(null); const usableCheckpoint = !!checkpoint && !checkpoint.startsWith("external::") && !looksLikePath(checkpoint); const needsCatalog = !usableCheckpoint; // Only when the checkpoint can't answer it: /v1/models scans the model dirs and - // HF caches. Re-runs when a model is loaded or unloaded. + // HF caches. Re-runs when a model is loaded or unloaded, and retries until the + // server reports something servable, since a download moves no store state. + // biome-ignore lint/correctness/useExhaustiveDependencies: a load or unload must refetch the servable ids useEffect(() => { if (!needsCatalog) return; let cancelled = false; - void listOpenAIModels() - .then((models) => { - if (!cancelled) setCatalog(models); - }) - .catch(() => { - // Best-effort: fall through to the placeholder. - }); + let timeoutId: number | null = null; + + const update = () => { + void listOpenAIModels() + .then((models) => { + if (!cancelled) setCatalog(models); + return models.length > 0; + }) + .catch(() => false) + .then((resolved) => { + if (cancelled || resolved) return; + timeoutId = window.setTimeout(update, CATALOG_RETRY_MS); + }); + }; + + update(); return () => { cancelled = true; + if (timeoutId !== null) window.clearTimeout(timeoutId); }; - }, [needsCatalog]); + }, [needsCatalog, checkpoint, ggufVariant]); return useMemo(() => { if (usableCheckpoint && checkpoint) { @@ -403,9 +419,9 @@ function useExampleModelName(): string { // No usable checkpoint (none, an external provider, or a raw load path): // name something this server actually holds, quant included so the request // pins the file on disk instead of letting the server pick a quant. - const pick = catalog.find((m) => m.loaded) ?? catalog[0]; + const pick = catalog?.find((m) => m.loaded) ?? catalog?.[0]; if (!pick) { - return MODEL_FALLBACK; + return null; } return pick.quant && !pick.id.includes(":") ? `${pick.id}:${pick.quant}` @@ -560,8 +576,9 @@ export function UsageExamples({ apiKey }: { apiKey?: string | null }) { const model = useExampleModelName(); const key = apiKey || KEY_PLACEHOLDER; + // Null model: nothing is servable, so there is no snippet worth copying. const snippets = useMemo( - () => buildSnippets(base, key, model, os), + () => (model ? buildSnippets(base, key, model, os) : null), [base, key, model, os], ); // Agent command must target the server the panel shows, not the :8888 default. @@ -580,6 +597,7 @@ export function UsageExamples({ apiKey }: { apiKey?: string | null }) { : "python"; const handleCopy = async () => { + if (!snippets) return; if (await copyToClipboard(snippets[lang])) { setCopied(true); setTimeout(() => setCopied(false), 1800); @@ -723,25 +741,33 @@ export function UsageExamples({ apiKey }: { apiKey?: string | null }) { ) : null} -
- + - {copied ? t("settings.apiKeys.copied") : t("settings.apiKeys.copy")} - - -
+ + ) : ( +
+ {t("settings.apiKeys.usageNoModel")} +
+ )}
{t("settings.apiKeys.codingAgents")} diff --git a/studio/frontend/src/i18n/locales/en.ts b/studio/frontend/src/i18n/locales/en.ts index bda564e4d7e..24bf9555376 100644 --- a/studio/frontend/src/i18n/locales/en.ts +++ b/studio/frontend/src/i18n/locales/en.ts @@ -722,6 +722,8 @@ export const en = { copyAccessToken: "Copy access token", copyNow: "Copy now - this won't be shown again.", usageExamples: "Usage examples", + usageNoModel: + "Load or download a model to see runnable examples. This server has no model to name yet.", usageTools: "Tools", exampleCurlTools: "curl + tools", examplePythonTools: "Python + tools", diff --git a/tests/studio/test_usage_examples_model_source_contract.py b/tests/studio/test_usage_examples_model_source_contract.py index 1224b93209a..8b93bbfcde4 100644 --- a/tests/studio/test_usage_examples_model_source_contract.py +++ b/tests/studio/test_usage_examples_model_source_contract.py @@ -4,6 +4,7 @@ """Static contract for which model the API usage examples name, and for the model-auto-switch control living in exactly one place on the API keys tab.""" +import re from pathlib import Path REPO = Path(__file__).resolve().parents[2] @@ -24,8 +25,8 @@ def test_examples_name_a_model_the_server_can_serve(): hook = src[src.find("function useExampleModelName") : src.find("// Backend PATH detection")] assert "listOpenAIModels()" in hook # Precedence: live checkpoint, then a loaded catalog entry, then any entry, - # and only then the placeholder. - assert "catalog.find((m) => m.loaded) ?? catalog[0]" in hook + # and only then no model at all. + assert "catalog?.find((m) => m.loaded) ?? catalog?.[0]" in hook # The snippet pins the quant so the request names the file on disk. assert "`${pick.id}:${pick.quant}`" in hook @@ -33,6 +34,42 @@ def test_examples_name_a_model_the_server_can_serve(): assert 'authFetch("/v1/models")' in api +def test_examples_never_print_a_hardcoded_model_id(): + # The bug this contract exists for. The catalog started as `[]`, so the very + # first render, and every render after a slow or failed /v1/models fetch, + # printed a copyable snippet naming a repo id the server cannot serve. The + # catalog is tri-state now (null until /v1 answers) and the panel says to + # load or download a model rather than naming one that does not exist. + src = USAGE_EXAMPLES_TSX.read_text(encoding = "utf-8") + assert "MODEL_FALLBACK" not in src + # No repo-shaped literal anywhere: a snippet may only name what /v1 returns. + assert re.search(r'"unsloth/[^"]+"', src) is None + assert "function useExampleModelName(): string | null" in src + assert "useState(null)" in src + # Nothing servable means nothing is built, so there is nothing to copy. + assert "(model ? buildSnippets(base, key, model, os) : null)" in src + assert "if (!snippets) return;" in src + assert "{snippets ? (" in src + assert 't("settings.apiKeys.usageNoModel")' in src + + en = EN_TS.read_text(encoding = "utf-8") + assert "usageNoModel:" in en + + +def test_catalog_refresh_follows_the_loaded_model(): + # `[needsCatalog]` alone never re-ran: it stays true the whole time there is + # no local checkpoint, so a model finishing its load left the snippet naming + # whatever the first fetch happened to see. + src = USAGE_EXAMPLES_TSX.read_text(encoding = "utf-8") + hook = src[src.find("function useExampleModelName") : src.find("// Backend PATH detection")] + assert "}, [needsCatalog, checkpoint, ggufVariant]);" in hook + # A finishing download moves no store state at all, so the fetch also + # retries itself until /v1 has something, on a timer the effect clears. + assert "window.setTimeout(update, CATALOG_RETRY_MS)" in hook + assert "window.clearTimeout(timeoutId)" in hook + assert "const CATALOG_RETRY_MS = 15000;" in src + + def test_usage_examples_has_no_duplicate_auto_switch_control(): # ModelAutoSwitchSection renders the same setting immediately below this # panel on the same tab, and the two do not share state, so a second switch From d76f56d372f5d8e1888aebbba58be1bb27e508cf Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Sun, 26 Jul 2026 12:44:20 +0000 Subject: [PATCH 13/56] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- studio/backend/tests/test_openai_auto_switch.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/studio/backend/tests/test_openai_auto_switch.py b/studio/backend/tests/test_openai_auto_switch.py index 394833a38bc..e3386e9a732 100644 --- a/studio/backend/tests/test_openai_auto_switch.py +++ b/studio/backend/tests/test_openai_auto_switch.py @@ -294,9 +294,9 @@ def test_openai_compat_routes_bound_to_handlers_with_auth(): for key, handler in expected.items(): assert key in seen, f"route {key} is not registered" route = seen[key] - assert route.endpoint.__name__ == handler, ( - f"{key} bound to {route.endpoint.__name__}, expected {handler}" - ) + assert ( + route.endpoint.__name__ == handler + ), f"{key} bound to {route.endpoint.__name__}, expected {handler}" deps = [d.call.__name__ for d in route.dependant.dependencies] assert "get_current_subject" in deps, f"{key} lost its auth dependency" From f913d33d56a3ee4cc3f2d24dddf6a112eaebe524 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 26 Jul 2026 13:09:18 +0000 Subject: [PATCH 14/56] Studio: scope the auto-download 404 cache to the caller's credentials The Hub answers 404 for a private repo the caller cannot see, so caching that verdict per repo alone let one anonymous request mark a private repo unservable for everyone for the whole TTL. A later caller sending a valid X-Unsloth-HF-Token skipped the probe and fell through to the resident model instead of downloading what it asked for. Keyed on the repo id plus a digest of the token now, so the token itself is never held. Two more from the same review: - Clear the chat runtime checkpoint after unloading from the API monitor, as the chat eject flow already does. The store went on treating the freed checkpoint as loaded and the usage examples kept naming it. - Point gated and not-found callers at the X-Unsloth-HF-Token header. Automatic download deliberately ignores the server's own Hugging Face identity, so telling the user to add a token in Studio sent them round the same 403 forever. --- .../core/inference/openai_auto_download.py | 37 ++++++++++++++----- .../tests/test_openai_auto_download.py | 33 +++++++++++++++++ .../components/api-monitor-console.tsx | 4 ++ 3 files changed, 64 insertions(+), 10 deletions(-) diff --git a/studio/backend/core/inference/openai_auto_download.py b/studio/backend/core/inference/openai_auto_download.py index a65bd70853b..bd6d0bce874 100644 --- a/studio/backend/core/inference/openai_auto_download.py +++ b/studio/backend/core/inference/openai_auto_download.py @@ -131,15 +131,28 @@ def looks_like_quant(variant: Optional[str]) -> bool: return _GGUF_KNOWN_QUANT_RE.fullmatch(variant.strip()) is not None -def _mark_not_servable(repo_id: str) -> None: +def _servable_key(repo_id: str, hf_token: Optional[str]) -> str: + """Cache key, per credential. + + The Hub answers 404 for a private repo the caller cannot see, so a verdict + reached without a token says nothing about a caller who has one. Keyed on a + digest so the token itself is never held here. + """ + import hashlib + + seen_as = hashlib.sha256(hf_token.encode()).hexdigest()[:16] if hf_token else "anon" + return f"{repo_id.lower()}\n{seen_as}" + + +def _mark_not_servable(repo_id: str, hf_token: Optional[str]) -> None: with _cache_lock: if len(_not_servable) >= _NOT_SERVABLE_MAX: _not_servable.clear() - _not_servable[repo_id.lower()] = time.monotonic() + _NOT_SERVABLE_TTL_S + _not_servable[_servable_key(repo_id, hf_token)] = time.monotonic() + _NOT_SERVABLE_TTL_S -def _is_not_servable(repo_id: str) -> bool: - key = repo_id.lower() +def _is_not_servable(repo_id: str, hf_token: Optional[str]) -> bool: + key = _servable_key(repo_id, hf_token) with _cache_lock: expires = _not_servable.get(key) if expires is None: @@ -155,8 +168,9 @@ def _gated_refusal(repo_id: str) -> AutoDownloadRefusal: status = 403, code = "model_access_denied", message = ( - f"'{repo_id}' is gated on Hugging Face. Accept its licence and add an " - "access token in Unsloth Studio, then retry." + f"'{repo_id}' is gated on Hugging Face. Accept its licence, then retry with " + "your own token in the X-Unsloth-HF-Token header: automatic download never " + "uses this server's Hugging Face identity." ), ) @@ -341,7 +355,7 @@ async def maybe_auto_download( repo_id, wanted_variant = split_model_ref(requested_model) if not is_downloadable_ref(requested_model): return None - if _is_not_servable(repo_id) and not looks_like_quant(wanted_variant): + if _is_not_servable(repo_id, hf_token) and not looks_like_quant(wanted_variant): return None # Adopt or reject against the single in-flight slot before touching the @@ -429,7 +443,7 @@ def _probe(): if status == 403: return _gated_refusal(repo_id) if status == 404: - _mark_not_servable(repo_id) + _mark_not_servable(repo_id, hf_token) # An id the Hub does not know is a foreign label, not a miss: pass it # to the resident model as before rather than 404 every LiteLLM and # OpenRouter "vendor/model". An explicit quant is a deliberate GGUF @@ -440,7 +454,10 @@ def _probe(): return AutoDownloadRefusal( status = 404, code = "model_not_found", - message = f"'{repo_id}' was not found on Hugging Face, or is not accessible.", + message = ( + f"'{repo_id}' was not found on Hugging Face, or is not accessible. " + "If it is private, send a token in the X-Unsloth-HF-Token header." + ), ) logger.warning("auto-download: Hub lookup failed for %r: %s", repo_id, exc) return AutoDownloadRefusal( @@ -460,7 +477,7 @@ def _probe(): variants = _gguf_variants(getattr(info, "siblings", None)) if not variants: _release(active) - _mark_not_servable(repo_id) + _mark_not_servable(repo_id, hf_token) if not looks_like_quant(wanted_variant): return None return AutoDownloadRefusal( diff --git a/studio/backend/tests/test_openai_auto_download.py b/studio/backend/tests/test_openai_auto_download.py index 8cfef1e85d6..33705b50653 100644 --- a/studio/backend/tests/test_openai_auto_download.py +++ b/studio/backend/tests/test_openai_auto_download.py @@ -293,6 +293,39 @@ def test_a_foreign_id_is_probed_once_then_cached(hub): assert hub["probes"] == 1 +def test_an_anonymous_404_does_not_silence_an_authorised_caller(hub): + # The Hub answers 404 for a private repo the caller cannot see, so caching + # that verdict globally would let one anonymous request hide a private repo + # from the token holder for the whole TTL. + hub["raise"] = _hub_error(_repo_not_found_error(), 404, "nope") + assert _run("myorg/private-GGUF") is None + assert hub["probes"] == 1 + + hub["raise"] = None + refusal = _run("myorg/private-GGUF", hf_token = "hf_caller_own") + assert hub["probes"] == 2 + assert refusal.code == "model_downloading" + + +def test_the_cache_is_per_token(hub): + hub["raise"] = _hub_error(_repo_not_found_error(), 404, "nope") + assert _run("myorg/private-GGUF", hf_token = "hf_a") is None + assert _run("myorg/private-GGUF", hf_token = "hf_a") is None + assert hub["probes"] == 1 + # A different credential gets its own verdict. + assert _run("myorg/private-GGUF", hf_token = "hf_b") is None + assert hub["probes"] == 2 + + +def test_the_gated_message_names_the_header_that_actually_works(hub): + # Auto-download never uses the server's ambient token, so pointing the + # caller at a Studio setting would loop them on the same 403. + hub["info"] = _Info(_gguf_repo_info().siblings, gated = "manual") + hub["auth_denied"] = True + refusal = _run("meta-llama/Llama-2-7b-hf") + assert "X-Unsloth-HF-Token" in refusal.message + + def test_gated_repo_is_403(hub): hub["raise"] = _hub_error(_gated_error(), 403, "gated") refusal = _run("meta-llama/Llama-2-7b-hf") diff --git a/studio/frontend/src/features/settings/components/api-monitor-console.tsx b/studio/frontend/src/features/settings/components/api-monitor-console.tsx index 27439399c35..77f1e2492dd 100644 --- a/studio/frontend/src/features/settings/components/api-monitor-console.tsx +++ b/studio/frontend/src/features/settings/components/api-monitor-console.tsx @@ -25,6 +25,7 @@ import { unloadModel, } from "../../chat/api/chat-api"; import { resolveInferenceCheckpointId } from "../../chat/lib/apply-inference-status-to-store"; +import { useChatRuntimeStore } from "../../chat/stores/chat-runtime-store"; import type { ApiMonitorEntry, ApiMonitorResponse } from "../../chat/types/api"; const API_INFERENCE_PREFIX_RE = /^\/api\/inference/; @@ -292,6 +293,9 @@ export function ApiMonitorConsole(): ReactElement { return; } await unloadModel({ model_path: checkpoint }); + // Same as the chat eject flow: the store still holds the checkpoint this + // just freed, and the usage examples would keep naming it. + useChatRuntimeStore.getState().clearCheckpoint(); setError(null); await loadMonitor(); } catch (err: unknown) { From 014e1fb588e87031cfd07c9e8925d116e80810d9 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 26 Jul 2026 13:27:31 +0000 Subject: [PATCH 15/56] Studio: tighten the comments added by this branch --- studio/backend/auth/authentication.py | 3 +- studio/backend/core/inference/api_monitor.py | 11 +-- .../backend/core/inference/llama_keepwarm.py | 4 +- .../core/inference/openai_auto_download.py | 60 +++++-------- studio/backend/routes/inference.py | 60 +++++-------- studio/backend/routes/settings.py | 3 +- studio/backend/tests/test_api_monitor.py | 6 +- studio/backend/tests/test_model_ids.py | 3 +- .../tests/test_openai_auto_download.py | 86 ++++++++----------- .../backend/tests/test_openai_auto_switch.py | 60 +++++-------- studio/backend/tests/test_openai_catalog.py | 12 +-- .../frontend/src/features/chat/types/api.ts | 4 +- .../settings/api/openai-auto-switch.ts | 4 +- .../features/settings/api/openai-models.ts | 8 +- .../components/api-monitor-console.tsx | 33 +++---- .../settings/components/usage-examples.tsx | 33 +++---- ...st_usage_examples_model_source_contract.py | 45 ++++------ 17 files changed, 164 insertions(+), 271 deletions(-) diff --git a/studio/backend/auth/authentication.py b/studio/backend/auth/authentication.py index cdb2b897c50..2481cd13e64 100644 --- a/studio/backend/auth/authentication.py +++ b/studio/backend/auth/authentication.py @@ -164,8 +164,7 @@ async def get_current_subject_allow_password_change( ) -# The literal the copyable API examples ship with. Pasting a snippet unedited is -# a much likelier mistake than a revoked key, so say so instead of "invalid". +# The literal the examples ship with; pasting one unedited is likelier than a revoked key. API_KEY_PLACEHOLDER = f"{API_KEY_PREFIX}YOUR_KEY" diff --git a/studio/backend/core/inference/api_monitor.py b/studio/backend/core/inference/api_monitor.py index d84de3ab2bc..a28210caf23 100644 --- a/studio/backend/core/inference/api_monitor.py +++ b/studio/backend/core/inference/api_monitor.py @@ -52,9 +52,8 @@ class ApiMonitorEntry: total_tokens: Optional[int] = None total_tokens_authoritative: bool = False error: Optional[str] = None - # "request" (an HTTP call) or "lifecycle" (a model load/unload). Lifecycle - # rows carry event/reason instead of a prompt and are server-wide, so they - # are shared across subjects rather than owned by the caller that caused them. + # "request" (an HTTP call) or "lifecycle" (a model load/unload). Lifecycle rows carry + # event/reason instead of a prompt, and are server-wide so every subject sees them. kind: str = "request" event: Optional[str] = None reason: Optional[str] = None @@ -304,8 +303,7 @@ def fail_open(self, entry_id: Optional[str], error: str) -> None: entry = self._find_locked(entry_id) if entry is None or entry.finished_at is not None: return - # Same lock as the check: a finish() landing in between would - # otherwise stamp this error onto a row that in fact succeeded. + # Same lock as the check, so a finish() cannot land in between. self._fail_locked(entry, error) def fail(self, entry_id: Optional[str], error: str) -> None: @@ -361,8 +359,7 @@ def get( return entry.snapshot(include_details = True) def active_count(self, *, subject: Optional[str] = None) -> int: - # Lifecycle rows are excluded: a load in progress is "running" so the row - # can show as live, but it is not an in-flight API request. + # Lifecycle rows show as "running" while loading but are not in-flight API requests. with self._lock: return sum( 1 diff --git a/studio/backend/core/inference/llama_keepwarm.py b/studio/backend/core/inference/llama_keepwarm.py index 44ce54d8ba5..88ea5360cc6 100644 --- a/studio/backend/core/inference/llama_keepwarm.py +++ b/studio/backend/core/inference/llama_keepwarm.py @@ -423,8 +423,8 @@ async def idle_unload_loop(poll_seconds: float = 15.0) -> None: elif manifest: _delete_resume_files(manifest) logger.info("Idle auto-unload: freed GGUF after %ss idle", ttl) - # This path deliberately skips note_model_unloaded (an idle - # unload stashes for reload), so record the monitor row here. + # This path skips note_model_unloaded (an idle unload stashes for + # reload), so record the monitor row here. _note_idle_unload_event(freed) seen_model = None except Exception as exc: diff --git a/studio/backend/core/inference/openai_auto_download.py b/studio/backend/core/inference/openai_auto_download.py index bd6d0bce874..e58a7a3a481 100644 --- a/studio/backend/core/inference/openai_auto_download.py +++ b/studio/backend/core/inference/openai_auto_download.py @@ -37,8 +37,7 @@ logger = get_logger(__name__) -# Hub metadata is one small request; keep it short so a slow Hub can't stall the -# request path for long. +# Keep the Hub probe short so a slow Hub can't stall the request path. _MODEL_INFO_TIMEOUT_S = 8.0 # Headroom left free after the download, so filling the disk can't wedge the box. _DISK_RESERVE_BYTES = 5 * 1024**3 @@ -73,9 +72,8 @@ class _Active: _lock = threading.Lock() _active: Optional[_Active] = None -# Repos the Hub says this server cannot serve. A LiteLLM or OpenRouter client -# names every provider "vendor/model", so without this each such request pays -# another Hub probe on the way to being answered by the resident model. +# Repos the Hub says this server cannot serve. Without it, every LiteLLM or OpenRouter +# "vendor/model" request pays a Hub probe before falling through to the resident model. _NOT_SERVABLE_TTL_S = 10 * 60 _NOT_SERVABLE_MAX = 256 _cache_lock = threading.Lock() @@ -248,8 +246,7 @@ async def _job_state(repo_id: str, variant: Optional[str]) -> tuple[str, Optiona status = await downloads.get_download_status_response(repo_id, variant or "") return status.state, status.error except Exception as exc: - # "unknown", not "idle": a probe that failed says nothing about the - # worker, and idle is read as "the job vanished" and ends the watch. + # "unknown", not "idle": idle ends the watch, and a failed probe proves nothing. logger.debug("auto-download: status probe failed for %r: %s", repo_id, exc) return "unknown", None @@ -301,8 +298,7 @@ async def _watch(active: _Active, hf_token: Optional[str]) -> None: await asyncio.sleep(_WATCH_POLL_S) state, error = await _job_state(active.repo_id, active.variant) if state in ("running", "cancelling", "unknown"): - # Only "running" has progress to report; the other two are still - # in flight, so keep the slot rather than failing the row. + # Only "running" has progress; the others are still in flight, so keep the slot. if state == "running": api_monitor.set_progress( active.monitor_id, @@ -312,8 +308,7 @@ async def _watch(active: _Active, hf_token: Optional[str]) -> None: ) continue if state == "complete": - # Drop the 5s resolver cache so the next retry resolves the new - # model instead of missing it again. + # Drop the resolver cache so the next retry sees the new model. await asyncio.to_thread(invalidate_index) api_monitor.finish(active.monitor_id, status = "completed") elif state == "idle": @@ -358,8 +353,8 @@ async def maybe_auto_download( if _is_not_servable(repo_id, hf_token) and not looks_like_quant(wanted_variant): return None - # Adopt or reject against the single in-flight slot before touching the - # network, so retries during a long download stay free. + # Adopt or reject against the single in-flight slot before touching the network, so + # retries during a long download stay free. with _lock: current = _active if current is not None and current.repo_id == repo_id: @@ -399,8 +394,7 @@ async def maybe_auto_download( if adopted.variant is None: # Another request is still probing this same repo. return _downloading_refusal(adopted.repo_id, None) - # complete/idle/cancelled: the watcher is about to release the slot, so - # ask for one more retry. Only "complete" actually got there. + # complete/idle/cancelled: the watcher is about to free the slot, so retry once more. return _downloading_refusal( _public_label(adopted.repo_id, adopted.variant), 100.0 if state == "complete" else None, @@ -411,11 +405,8 @@ async def maybe_auto_download( repo_id, wanted_variant, requested_model, hf_token, provisional ) except BaseException: - # BaseException, not Exception: asyncio.CancelledError has not been an - # Exception since 3.8, so a request cancelled mid-probe would otherwise - # leave the provisional slot installed with variant None -- every other - # repo permanently model_download_busy and this one permanently - # "downloading", until the process restarts. + # BaseException, not Exception: CancelledError is not an Exception, so a request + # cancelled mid-probe would otherwise wedge the provisional slot until restart. _release(provisional) raise @@ -444,10 +435,8 @@ def _probe(): return _gated_refusal(repo_id) if status == 404: _mark_not_servable(repo_id, hf_token) - # An id the Hub does not know is a foreign label, not a miss: pass it - # to the resident model as before rather than 404 every LiteLLM and - # OpenRouter "vendor/model". An explicit quant is a deliberate GGUF - # reference, so that one is answered. + # An id the Hub does not know is a foreign label, not a miss: fall through to the + # resident model. An explicit quant is a deliberate GGUF reference, so answer it. if not looks_like_quant(wanted_variant): return None # A private repo reads as absent without a token; don't confirm either way. @@ -468,9 +457,8 @@ def _probe(): ) if getattr(info, "gated", False) and await asyncio.to_thread(_auth_denied, repo_id, hf_token): - # The Hub serves metadata for a gated repo without granting its files, so - # a probe that returned is not proof of access. Left unchecked the config - # read below fails instead and reports the unrelated custom-code refusal. + # The Hub serves metadata for a gated repo without granting its files, so a probe that + # returned is not proof of access. Unchecked, the config read below misreports it. _release(active) return _gated_refusal(repo_id) @@ -489,9 +477,8 @@ def _probe(): ), ) - # trust_remote_code gate. _config_has_auto_map is tri-state: refuse on True - # and on None (unreadable), rather than _requires_trust_remote_code_for_model, - # which swallows errors into False. Fine as a UI hint, wrong as an admission. + # trust_remote_code gate. _config_has_auto_map is tri-state, so refuse on True and on None + # (unreadable); _requires_trust_remote_code_for_model swallows errors into False. from utils.security.consent import _config_has_auto_map has_auto_map = await asyncio.to_thread(_config_has_auto_map, repo_id, hf_token) @@ -555,8 +542,7 @@ def _match_variant(wanted: Optional[str], variants: dict[str, int]) -> Optional[ return lowered.get(wanted.strip().lower()) from utils.models.model_config import _pick_best_gguf - # _pick_best_gguf ranks filenames by quant substring, so give it synthesized - # "