Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
102 changes: 27 additions & 75 deletions studio/backend/tests/test_install_resolve_prebuilt.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,9 +24,7 @@

ilp = importlib.import_module("install_llama_prebuilt")

if not hasattr(ilp, "published_repo_for_host") or not hasattr(
ilp, "resolve_simple_install_release_plans"
):
if not hasattr(ilp, "resolve_simple_install_release_plans"):
pytest.skip("PR symbols not present - check branch", allow_module_level = True)

FORK = ilp.DEFAULT_PUBLISHED_REPO # unslothai/llama.cpp
Expand Down Expand Up @@ -56,73 +54,6 @@ def _host(**kw):
return ilp.HostInfo(**base)


def test_published_repo_for_host():
# CPU-only Linux (x64 and arm64) -> ggml-org upstream.
assert ilp.published_repo_for_host(_host(is_linux = True, is_x86_64 = True)) == UPSTREAM
assert (
ilp.published_repo_for_host(_host(is_linux = True, is_arm64 = True, machine = "aarch64"))
== UPSTREAM
)
# GPU Linux -> fork.
assert (
ilp.published_repo_for_host(_host(is_linux = True, is_x86_64 = True, has_usable_nvidia = True))
== FORK
)
assert ilp.published_repo_for_host(_host(is_linux = True, is_x86_64 = True, has_rocm = True)) == FORK
# CPU-only Windows -> ggml-org (setup.ps1: the fork ships no win-cpu bundle).
assert (
ilp.published_repo_for_host(_host(system = "Windows", is_windows = True, is_x86_64 = True))
== UPSTREAM
)
# GPU Windows -> fork.
assert (
ilp.published_repo_for_host(
_host(system = "Windows", is_windows = True, is_x86_64 = True, has_usable_nvidia = True)
)
== FORK
)
# macOS -> fork regardless of GPU (ggml-org macOS bundles need too-new macOS).
assert (
ilp.published_repo_for_host(
_host(system = "Darwin", is_macos = True, is_arm64 = True, machine = "arm64")
)
== FORK
)
# Linux with AMD tooling but no probed GPU -> fork (setup.sh routes on tooling).
assert (
ilp.published_repo_for_host(
_host(is_linux = True, is_x86_64 = True), linux_amd_tooling_present = True
)
== FORK
)
# The tooling hint is Linux-only: Windows CPU stays on ggml-org.
assert (
ilp.published_repo_for_host(
_host(system = "Windows", is_windows = True, is_x86_64 = True),
linux_amd_tooling_present = True,
)
== UPSTREAM
)


def test_macos_intel_and_arm_both_route_to_fork():
# macOS uses the unslothai fork's own Mac prebuilts for BOTH arm64 and Intel;
# there is no longer any upstream-on-macOS default path, so the obsolete
# pre-macOS-26 pin (b9415) is gone.
assert (
ilp.published_repo_for_host(
_host(system = "Darwin", is_macos = True, is_arm64 = True, machine = "arm64")
)
== FORK
)
assert (
ilp.published_repo_for_host(
_host(system = "Darwin", is_macos = True, is_x86_64 = True, machine = "x86_64")
)
== FORK
)


def test_macos_upstream_pin_only_for_explicit_pre26_upstream():
pre26 = _host(
system = "Darwin",
Expand Down Expand Up @@ -188,15 +119,13 @@ def test_resolve_prebuilt_unavailable(monkeypatch, capsys):
assert out["repo"] == FORK


def test_resolve_prebuilt_linux_amd_tooling_routes_to_fork(monkeypatch, capsys):
# CPU-probed Linux host but rocminfo on PATH: the dispatch must route to the
# fork so a HIP source build is not offered an upstream CPU prebuilt.
monkeypatch.setattr(ilp, "detect_host", lambda: _host(is_linux = True, is_x86_64 = True))
monkeypatch.setattr(ilp.shutil, "which", lambda tool: tool == "rocminfo")
def _run_resolve_capture_host(monkeypatch, capsys):
"""Drive --resolve-prebuilt and return the host the resolver was handed."""
seen = {}

def _resolver(tag, host, repo, published_release_tag):
seen["repo"] = repo
seen["host"] = host
raise ilp.PrebuiltFallback("no asset")

monkeypatch.setattr(ilp, "resolve_simple_install_release_plans", _resolver)
Expand All @@ -207,10 +136,33 @@ def _resolver(tag, host, repo, published_release_tag):
)
assert ilp.main() == ilp.EXIT_SUCCESS
out = json.loads(capsys.readouterr().out.strip().splitlines()[-1])
return seen, out


def test_resolve_prebuilt_cpu_linux_routes_to_fork(monkeypatch, capsys):
# CPU-only Linux host (no GPU): the dispatch routes to the fork, which now
# ships the CPU prebuilt -- it no longer falls back to ggml-org upstream.
monkeypatch.setattr(ilp, "detect_host", lambda: _host(is_linux = True, is_x86_64 = True))
seen, out = _run_resolve_capture_host(monkeypatch, capsys)
assert seen["repo"] == FORK
assert out["repo"] == FORK


def test_resolve_prebuilt_rocm_sdk_only_host_still_offered_cpu(monkeypatch, capsys):
# A CPU-only host that merely has ROCm/HIP SDK tools on PATH (no AMD GPU, so
# detect_host leaves has_rocm False) is a valid CPU-prebuilt target. The probe
# must NOT reclassify it as ROCm from tool presence alone and suppress the CPU
# bundle -- that would deny the fork CPU prebuilt to a legitimate CPU source
# build. The host is left CPU-only and resolves against the fork.
monkeypatch.setattr(ilp, "detect_host", lambda: _host(is_linux = True, is_x86_64 = True))
monkeypatch.setattr(
ilp.shutil, "which", lambda tool: "/opt/rocm/bin/hipconfig" if tool == "hipconfig" else None
)
seen, out = _run_resolve_capture_host(monkeypatch, capsys)
assert seen["repo"] == FORK
assert seen["host"].has_rocm is False


# Blackwell floor is sm_100 (data-center B100/B200, B300/GB300), below consumer
# sm_120 -- 120 wrongly excluded data-center hosts from the prebuilt selection.

Expand Down
63 changes: 24 additions & 39 deletions studio/install_llama_prebuilt.py
Original file line number Diff line number Diff line change
Expand Up @@ -165,9 +165,9 @@ def env_int(
# errors. Only use "master" temporarily when the latest release is missing
# support for a new model architecture.
DEFAULT_LLAMA_TAG = os.environ.get("UNSLOTH_LLAMA_TAG", "latest")
# Default published repo for prebuilt release resolution. Linux uses
# Unsloth prebuilts; setup.sh/setup.ps1 pass --published-repo explicitly
# for macOS/Windows to override with ggml-org/llama.cpp when needed.
# Default published repo for prebuilt release resolution. Every host plans
# its prebuilt against the Unsloth fork; setup.sh/setup.ps1 pass it via
# --published-repo. ggml-org is reachable only via an explicit override.
DEFAULT_PUBLISHED_REPO = "unslothai/llama.cpp"
DEFAULT_PUBLISHED_TAG = os.environ.get("UNSLOTH_LLAMA_RELEASE_TAG")
DEFAULT_PUBLISHED_MANIFEST_ASSET = os.environ.get(
Expand Down Expand Up @@ -3135,21 +3135,6 @@ def _apply_host_overrides(
return host


def published_repo_for_host(host: HostInfo, *, linux_amd_tooling_present: bool = False) -> str:
"""The release repo setup.sh / setup.ps1 pick for this host: macOS always the
fork (ggml-org macOS bundles need too-new macOS); else CPU-only Linux/Windows
-> ggml-org upstream (the fork ships no CPU bundle) and any usable GPU (NVIDIA
or ROCm) -> the fork. linux_amd_tooling_present mirrors setup.sh routing Linux
hosts that expose AMD tooling (rocminfo/amd-smi/hipconfig/hipinfo) to the fork
even when the probe cannot confirm an active GPU. Mirrors the shell routing."""
if host.is_macos:
return DEFAULT_PUBLISHED_REPO
has_gpu = (
host.has_usable_nvidia or host.has_rocm or (host.is_linux and linux_amd_tooling_present)
)
return DEFAULT_PUBLISHED_REPO if has_gpu else UPSTREAM_REPO


def pick_windows_cuda_runtime(host: HostInfo) -> str | None:
if not host.driver_cuda_version:
return None
Expand Down Expand Up @@ -4015,6 +4000,9 @@ def resolve_release_asset_choice(
published_choice = published_rocm_choice_for_host(release, host, "windows-rocm")
else:
published_choice = published_asset_choice_for_kind(release, "windows-cpu")
elif host.is_windows and host.is_arm64:
# Windows arm64 has no GPU prebuilt, so it always takes the CPU bundle.
published_choice = published_asset_choice_for_kind(release, "windows-arm64")
elif host.is_macos and host.is_arm64:
published_choice = published_asset_choice_for_kind(release, "macos-arm64")
elif host.is_macos and host.is_x86_64:
Expand Down Expand Up @@ -6127,8 +6115,13 @@ def _linux_published_attempts(host: HostInfo, bundle: PublishedReleaseBundle) ->
# CPU-only host. A usable-NVIDIA host never reaches here -- if its CUDA
# selection produced nothing we want an empty attempt list so the caller
# source-builds with CUDA, not a CPU-only binary silently installed on a
# GPU host (mirrors the ROCm branch, and Windows NVIDIA).
cpu_choice = published_asset_choice_for_kind(bundle, "linux-cpu")
# GPU host (mirrors the ROCm branch, and Windows NVIDIA). Only x86_64 and
# arm64 have a CPU bundle; any other Linux arch (ppc64le, riscv64, s390x)
# has none, so leave attempts empty and source-build rather than hand it
# the x86_64 linux-cpu binary (the Linux preflight checks libraries, not
# ELF arch, so a wrong-arch binary would not be caught).
kind = "linux-cpu" if host.is_x86_64 else "linux-arm64" if host.is_arm64 else None
cpu_choice = published_asset_choice_for_kind(bundle, kind) if kind else None
if cpu_choice is not None:
attempts.append(cpu_choice)
return attempts
Expand All @@ -6143,9 +6136,9 @@ def _fork_manifest_release_plans(
max_release_fallbacks: int = DEFAULT_MAX_PREBUILT_RELEASE_FALLBACKS,
) -> tuple[str, list[InstallReleasePlan]]:
"""Manifest-reading branch of resolve_simple_install_release_plans, used for
the fork's bundles whose GPU/arch coverage lives in
llama-prebuilt-manifest.json rather than in the filename: arm64 CUDA, Windows
CUDA, per-gfx ROCm, and macOS. Linux x64 takes the faster filename path."""
every fork host: all of the fork's bundles describe their GPU/arch coverage
in llama-prebuilt-manifest.json rather than in the asset filename (CPU,
x64/arm64 CUDA, Windows CUDA, per-gfx ROCm, and macOS)."""
requested_tag = normalized_requested_llama_tag(llama_tag)
allow_older_release_fallback = requested_tag == "latest" and not published_release_tag
release_limit = max(1, max_release_fallbacks)
Expand Down Expand Up @@ -6714,8 +6707,8 @@ def install_prebuilt(
log(
f"no existing llama.cpp install detected at {install_dir}; performing fresh prebuilt install"
)
# Single resolver: linux-x64 takes the fast filename path internally,
# every other fork host reads the manifest.
# Single resolver: every fork host selects from the release manifest;
# an explicit ggml-org override selects by asset filename instead.
requested_tag, release_plans = resolve_simple_install_release_plans(
llama_tag,
host,
Expand Down Expand Up @@ -6903,8 +6896,8 @@ def parse_args() -> argparse.Namespace:
const = "latest",
help = (
"Report whether an official prebuilt exists for this host without "
"downloading. Picks the host's published repo when --published-repo "
"is left at the default. Use --output-format json."
"downloading. Plans against --published-repo (defaults to the "
"fork). Use --output-format json."
),
)
parser.add_argument(
Expand Down Expand Up @@ -6992,24 +6985,16 @@ def main() -> int:
return EXIT_SUCCESS

if args.resolve_prebuilt is not None:
# Host-aware "is a prebuilt available" probe, no download. A default repo
# means "pick the repo for this host"; PrebuiltFallback == source build.
# Host-aware "is a prebuilt available" probe, no download. Every host now
# plans against the fork (args.published_repo defaults to it); an explicit
# --published-repo overrides. PrebuiltFallback == source build.
host = _apply_host_overrides(
detect_host(),
override_has_rocm = args.has_rocm,
override_rocm_gfx = args.rocm_gfx,
force_cpu = args.cpu_fallback,
)
# setup.sh routes Linux hosts with AMD tooling to the fork even when no GPU
# is probed; mirror that so a HIP source build is not offered a CPU prebuilt.
amd_tooling = host.is_linux and any(
shutil.which(t) for t in ("rocminfo", "amd-smi", "hipconfig", "hipinfo")
)
repo = (
published_repo_for_host(host, linux_amd_tooling_present = amd_tooling)
if args.published_repo == DEFAULT_PUBLISHED_REPO
else args.published_repo
)
repo = args.published_repo
try:
_requested, plans = resolve_simple_install_release_plans(
args.resolve_prebuilt, host, repo, args.published_release_tag or ""
Expand Down
23 changes: 12 additions & 11 deletions studio/setup.ps1
Original file line number Diff line number Diff line change
Expand Up @@ -3088,12 +3088,11 @@ $LlamaCppDir = Join-Path $UnslothHome "llama.cpp"
$NeedLlamaSourceBuild = $false
$SkipPrebuiltInstall = $false
$RequestedLlamaTag = if ($env:UNSLOTH_LLAMA_TAG) { $env:UNSLOTH_LLAMA_TAG } else { $DefaultLlamaTag }
# GPU Windows (CUDA / ROCm) installs the fork's app-* prebuilts; CPU-only stays
# on ggml-org (the fork ships no windows-cpu bundle). Mirrors setup.sh's routing.
# A resolved gfx arch counts as a GPU host even when $HasROCm is false (Adrenalin
# driver only, no HIP runtime): the fork's per-gfx bundle ships its own runtime,
# so route there instead of ggml-org / a CPU build.
$HelperReleaseRepo = if ($HasNvidiaSmi -or $HasROCm -or $script:ROCmGfxArch) { "unslothai/llama.cpp" } else { "ggml-org/llama.cpp" }
# Every host installs the fork's app-* prebuilts now: GPU Windows (CUDA / ROCm)
# already did, and the fork now also ships the CPU bundles for Windows x64 and
# arm64 (windows-cpu / windows-arm64). ggml-org artifacts are no longer used by
# default. Mirrors setup.sh's routing.
$HelperReleaseRepo = "unslothai/llama.cpp"
$LlamaPr = if ($env:UNSLOTH_LLAMA_PR) { $env:UNSLOTH_LLAMA_PR.Trim() } else { "" }

$LlamaPrForce = if ($env:UNSLOTH_LLAMA_PR_FORCE) { $env:UNSLOTH_LLAMA_PR_FORCE.Trim() } else { $DefaultLlamaPrForce }
Expand Down Expand Up @@ -3283,11 +3282,13 @@ if ($LocalLlamaCppLinked) {
# treat a valid ROCm install as mismatched. A name-inferred gfx
# arch (Adrenalin-only, no confirmed runtime) still counts as
# ROCm-capable -- the ROCm prebuilt bundles its own runtime,
# mirroring the --rocm-gfx forward below. NOTE: this block is
# currently inert -- write_prebuilt_metadata does not persist an
# install_kind key, so $existingKind is always null. If that changes,
# add the remaining host kinds (e.g. windows-arm64) before relying on it.
$expectedKinds = if ($HasROCm -or $script:ROCmGfxArch) { @("windows-rocm", "windows-hip") } elseif ($HasNvidiaSmi) { @("windows-cuda") } else { @("windows-cpu") }
# mirroring the --rocm-gfx forward below. The CPU branch covers both
# the x64 windows-cpu and arm64 windows-arm64 bundles (Windows arm64
# has no GPU prebuilt). NOTE: this block is currently inert --
# write_prebuilt_metadata does not persist an install_kind key, so
# $existingKind is always null; keep $expectedKinds in sync with the
# kinds install_llama_prebuilt.py installs before relying on it.
$expectedKinds = if ($HasROCm -or $script:ROCmGfxArch) { @("windows-rocm", "windows-hip") } elseif ($HasNvidiaSmi) { @("windows-cuda") } else { @("windows-cpu", "windows-arm64") }
if ($existingKind -and ($existingKind -notin $expectedKinds)) {
substep "Removing mismatched llama.cpp install (found '$existingKind', need one of: $($expectedKinds -join ', '))..."
Remove-Item -Recurse -Force -LiteralPath $LlamaCppDir -ErrorAction SilentlyContinue
Expand Down
Loading
Loading