From 9376c9064b4ec0ffeec6e5a95048aed03e1e3a18 Mon Sep 17 00:00:00 2001 From: majiayu000 <1835304752@qq.com> Date: Mon, 29 Dec 2025 13:45:07 +0800 Subject: [PATCH 01/20] fix: add revision parameter support and escape quotes in chat templates - Fix #3544: Add revision parameter to AutoConfig, AutoModelForCausalLM, AutoModelForSequenceClassification, and load_correct_tokenizer calls in FastLlamaModel.from_pretrained. This enables loading specific model revisions/branches from HuggingFace Hub. - Fix #3667: Escape single quotes in system messages before substituting into Jinja2 templates. This prevents TemplateSyntaxError when system messages contain apostrophes (e.g., "user's" in Vicuna templates). Signed-off-by: majiayu000 <1835304752@qq.com> (cherry picked from commit b0a6e4154b1bca9ed9bc06bdd1a83da1007dd6bf) --- unsloth/chat_templates.py | 8 ++++++-- unsloth/models/llama.py | 4 ++++ unsloth/tokenizer_utils.py | 5 +++++ 3 files changed, 15 insertions(+), 2 deletions(-) diff --git a/unsloth/chat_templates.py b/unsloth/chat_templates.py index 35eb871529..50e7a1f380 100644 --- a/unsloth/chat_templates.py +++ b/unsloth/chat_templates.py @@ -1667,14 +1667,18 @@ def _change_system_message(template: str, type_chat_template: str, system_messag if has_placeholder: if system_message is None: raise ValueError("Unsloth: You need to provide a system message for custom templates.") - new_template = re.sub(system_message_pattern, system_message, template) + # Escape single quotes to prevent Jinja2 template syntax errors + escaped_message = system_message.replace("'", "\\'") + new_template = re.sub(system_message_pattern, escaped_message, template) return new_template, system_message return template, system_message # For predefined templates with default system message message_to_use = system_message if system_message is not None else default_system_message - new_template = re.sub(system_message_pattern, message_to_use, template) + # Escape single quotes to prevent Jinja2 template syntax errors + escaped_message = message_to_use.replace("'", "\\'") + new_template = re.sub(system_message_pattern, escaped_message, template) return new_template, message_to_use diff --git a/unsloth/models/llama.py b/unsloth/models/llama.py index 93d93e26d6..0498c11734 100644 --- a/unsloth/models/llama.py +++ b/unsloth/models/llama.py @@ -2331,6 +2331,7 @@ def from_pretrained( model_name, token = token, attn_implementation = "sdpa", + revision = revision, ) model_config.model_name = model_name model_max_seq_length = model_config.max_position_embeddings @@ -2424,6 +2425,7 @@ def from_pretrained( max_position_embeddings = max_position_embeddings, trust_remote_code = trust_remote_code, attn_implementation = preferred_attn_impl, + revision = revision, **kwargs, ) elif not fast_inference: @@ -2436,6 +2438,7 @@ def from_pretrained( max_position_embeddings = max_position_embeddings, trust_remote_code = trust_remote_code, attn_implementation = preferred_attn_impl, + revision = revision, **kwargs, ) model.fast_generate = make_fast_generate_wrapper(model.generate) @@ -2505,6 +2508,7 @@ def from_pretrained( token = token, trust_remote_code = trust_remote_code, fix_tokenizer = fix_tokenizer, + revision = revision, ) model, tokenizer = patch_tokenizer(model, tokenizer) diff --git a/unsloth/tokenizer_utils.py b/unsloth/tokenizer_utils.py index c445879df7..ccd0249209 100644 --- a/unsloth/tokenizer_utils.py +++ b/unsloth/tokenizer_utils.py @@ -503,6 +503,7 @@ def _load_correct_tokenizer( trust_remote_code = False, cache_dir = "huggingface_tokenizers_cache", fix_tokenizer = True, + revision = None, ): if IS_COLAB_ENVIRONMENT: cache_dir = cache_dir @@ -528,6 +529,7 @@ def _load_correct_tokenizer( legacy = False, from_slow = True, cache_dir = cache_dir, + revision = revision, ) except: slow_tokenizer = None @@ -546,6 +548,7 @@ def _load_correct_tokenizer( token = token, trust_remote_code = trust_remote_code, cache_dir = cache_dir, + revision = revision, ) if not fix_tokenizer or tokenizer_name in IGNORED_TOKENIZER_NAMES: @@ -587,6 +590,7 @@ def load_correct_tokenizer( trust_remote_code = False, cache_dir = "huggingface_tokenizers_cache", fix_tokenizer = True, + revision = None, ): tokenizer = _load_correct_tokenizer( tokenizer_name = tokenizer_name, @@ -596,6 +600,7 @@ def load_correct_tokenizer( trust_remote_code = trust_remote_code, cache_dir = cache_dir, fix_tokenizer = fix_tokenizer, + revision = revision, ) ### 1. Fixup tokenizer's chat_template From c823e52b2d9a308443e78795002e08b887f9b837 Mon Sep 17 00:00:00 2001 From: majiayu000 <1835304752@qq.com> Date: Mon, 29 Dec 2025 19:07:51 +0800 Subject: [PATCH 02/20] fix: propagate revision parameter to vLLM and PEFT loaders MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Add revision to load_vllm_kwargs in llama.py to fix config/weights mismatch - Add revision to PEFT AutoConfig calls in loader.py (FastLanguageModel & FastModel) Addresses reviewer feedback from @chatgpt-codex-connector and @Datta0 🤖 Generated with [Claude Code](https://claude.com/claude-code) Co-Authored-By: Claude Opus 4.5 (cherry picked from commit 14f89e453173f3279c885ed37d5f2a9cf10793df) --- unsloth/models/llama.py | 1 + unsloth/models/loader.py | 2 ++ 2 files changed, 3 insertions(+) diff --git a/unsloth/models/llama.py b/unsloth/models/llama.py index 0498c11734..fd2b3044c0 100644 --- a/unsloth/models/llama.py +++ b/unsloth/models/llama.py @@ -2471,6 +2471,7 @@ def from_pretrained( disable_log_stats = disable_log_stats, use_bitsandbytes = load_in_4bit, unsloth_vllm_standby = unsloth_vllm_standby, + revision = revision, fp8_mode = fp8_mode, ) for allowed_arg in allowed_args: diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index bd15ed5281..3155cd0d09 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -553,6 +553,7 @@ def from_pretrained( model_config = AutoConfig.from_pretrained( model_name, token = token, + revision = revision, trust_remote_code = trust_remote_code, ) @@ -1304,6 +1305,7 @@ def from_pretrained( model_config = AutoConfig.from_pretrained( model_name, token = token, + revision = revision, trust_remote_code = trust_remote_code, ) From ae38c3639de6965e24600e07eddc467dae15c332 Mon Sep 17 00:00:00 2001 From: majiayu000 <1835304752@qq.com> Date: Tue, 30 Dec 2025 14:16:23 +0800 Subject: [PATCH 03/20] fix: add revision parameter to FastBaseModel in vision.py MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Propagate revision parameter to all from_pretrained calls in vision.py to ensure consistent version pinning for vision models. 🤖 Generated with [Claude Code](https://claude.com/claude-code) Co-Authored-By: Claude Sonnet 4.5 (cherry picked from commit c5aa4ec92783345eaa0286bd876386cb2767d990) --- unsloth/models/vision.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/unsloth/models/vision.py b/unsloth/models/vision.py index a8adba99e7..835bbcb850 100644 --- a/unsloth/models/vision.py +++ b/unsloth/models/vision.py @@ -421,6 +421,7 @@ def from_pretrained( auto_config = None, offload_embedding = False, float32_mixed_precision = None, # Forces float32 mixed precision + revision = None, # vLLM parameters fast_inference = False, gpu_memory_utilization = 0.5, @@ -720,6 +721,7 @@ def from_pretrained( model_name, token = token, trust_remote_code = trust_remote_code, + revision = revision, ) if hasattr(auto_config, "quantization_config"): from transformers.quantizers.auto import ( @@ -776,12 +778,12 @@ def from_pretrained( model_name, token = token, trust_remote_code = trust_remote_code, + revision = revision, ) setattr(auto_config, "_attn_implementation", config_attn_impl) if hasattr(auto_config, "attn_implementation"): setattr(auto_config, "attn_implementation", config_attn_impl) model_config = auto_config - verify_fp8_support_if_applicable(model_config) raise_handler = RaiseUninitialized() @@ -796,6 +798,7 @@ def from_pretrained( # quantization_config = bnb_config, token = token, trust_remote_code = trust_remote_code, + revision = revision, # attn_implementation = attn_implementation, **kwargs, ) From 05f600e53153a02241b22bfd0a3be7085e0c0201 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 2 Aug 2026 14:50:56 +0000 Subject: [PATCH 04/20] Forward revision to the config, weight and tokenizer loads FastLlamaModel.from_pretrained took a `revision` argument and never read it, so the config, the weights and the tokenizer all came from the repo's default branch while the caller believed they had pinned a ref. Reported in #3544 by someone versioning their fine-tunes with branches, which makes it a silently wrong base checkpoint rather than an error. Forward it in llama.py (both AutoConfig loads, the three model loads, the tokenizer, the prefetch warm and the fp8 scale restore), plumb it through load_correct_tokenizer, and read it from kwargs in vision.py for the four AutoConfig, two processor and two tokenizer loads plus the VLM processor fallback. vision.py must not bind it as a named parameter: the weight load there forwards **kwargs, so binding it would drop it from that load. model_name is not always the repo the caller named. get_model_name can swap in a pre-quantized mirror, _offline_quantize_to_fp8 an fp8 temp dir, ModelScope a local snapshot, and fast_inference_setup a -bnb-4bit variant, and use_exact_model_name only gates the first of those. A ref from the original repo does not exist on the substitute, so _revision_for_resolved_repo drops it with a warning naming both repos when the resolution changed the name. The adapter load keeps the caller's revision, since that one really is for old_model_name. Supersedes the earlier attempt on this branch, whose chat-template hunk is handled by #7731 and #7746, whose vision.py signature change caused the drop described above, and whose load_vllm(revision = ...) raised TypeError because load_vllm has no such parameter. Fixes #3544 --- tests/python/test_revision_forwarding.py | 215 +++++++++++++++++++++++ unsloth/models/llama.py | 21 ++- unsloth/models/loader.py | 29 ++- unsloth/models/loader_utils.py | 2 + unsloth/models/vision.py | 28 ++- unsloth/tokenizer_utils.py | 5 + 6 files changed, 291 insertions(+), 9 deletions(-) create mode 100644 tests/python/test_revision_forwarding.py diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py new file mode 100644 index 0000000000..fcaffd11f9 --- /dev/null +++ b/tests/python/test_revision_forwarding.py @@ -0,0 +1,215 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. +"""`revision` must reach the config, weight and tokenizer loads (issue #3544). + +FastLlamaModel.from_pretrained took a `revision` argument and never read it, so the +config, weights and tokenizer silently came from the repo's default branch. These are +AST-structural so they need no GPU, no network and no gated checkpoint; importing +unsloth on a CPU runner is what tests/conftest.py exists to work around. +""" + +import ast +import types +from pathlib import Path + +import pytest + + +REPO = Path(__file__).parents[2] +LLAMA = REPO / "unsloth" / "models" / "llama.py" +LOADER = REPO / "unsloth" / "models" / "loader.py" +VISION = REPO / "unsloth" / "models" / "vision.py" +TOKENIZER_UTILS = REPO / "unsloth" / "tokenizer_utils.py" + + +def _tree(path): + return ast.parse(path.read_text(encoding = "utf-8")) + + +def _function(tree, name, class_name = None): + body = tree.body + if class_name is not None: + classes = [n for n in body if isinstance(n, ast.ClassDef) and n.name == class_name] + assert classes, f"{class_name} not found" + body = classes[0].body + for node in body: + if isinstance(node, ast.FunctionDef) and node.name == name: + return node + raise AssertionError(f"{class_name or ''}.{name} not found") + + +def _params(function): + return [a.arg for a in function.args.args + function.args.kwonlyargs] + + +def _calls(function, callee): + """Every Call whose dotted name ends with `callee`.""" + return [ + node for node in ast.walk(function) + if isinstance(node, ast.Call) and ast.unparse(node.func).split(".")[-1] == callee + ] + + +def _revision_kwarg(call): + for keyword in call.keywords: + if keyword.arg == "revision": + return keyword + return None + + +def test_fast_llama_model_reads_its_revision_argument(): + """The whole of #3544: the parameter existed but had zero reads.""" + function = _function(_tree(LLAMA), "from_pretrained", "FastLlamaModel") + assert "revision" in _params(function) + loads = [n for n in ast.walk(function) if isinstance(n, ast.Name) and n.id == "revision" + and isinstance(n.ctx, ast.Load)] + assert loads, "revision is accepted but never read" + + +@pytest.mark.parametrize( + "callee, minimum", + [ + ("AutoConfig", 2), # checkpoint probe + main config + ("AutoModelForCausalLM", 2), # user-config and plain branches + ("AutoModelForSequenceClassification", 1), + ("load_correct_tokenizer", 1), + ], +) +def test_llama_loads_forward_revision(callee, minimum): + function = _function(_tree(LLAMA), "from_pretrained", "FastLlamaModel") + calls = _calls(function, "from_pretrained") if callee != "load_correct_tokenizer" else \ + _calls(function, "load_correct_tokenizer") + if callee != "load_correct_tokenizer": + calls = [c for c in calls if ast.unparse(c.func).startswith(callee)] + assert len(calls) >= minimum, f"expected >= {minimum} {callee} loads, found {len(calls)}" + for call in calls: + assert _revision_kwarg(call) is not None, f"{callee} at line {call.lineno} drops revision" + + +def test_llama_does_not_pass_revision_to_load_vllm(): + """load_vllm has no `revision` parameter and load_vllm_kwargs is not filtered, + so putting one in that dict is an unconditional TypeError on the vLLM path.""" + function = _function(_tree(LLAMA), "from_pretrained", "FastLlamaModel") + dicts = [ + node.value for node in ast.walk(function) + if isinstance(node, ast.Assign) + and any(getattr(t, "id", None) == "load_vllm_kwargs" for t in node.targets) + and isinstance(node.value, ast.Call) + ] + assert dicts, "load_vllm_kwargs assignment not found" + for call in dicts: + assert _revision_kwarg(call) is None, "revision is not a load_vllm argument" + + +def test_fast_base_model_does_not_bind_revision(): + """vision.py's weight load forwards **kwargs, so binding `revision` as a named + parameter would silently drop it from there and from kwargs.get('revision').""" + function = _function(_tree(VISION), "from_pretrained", "FastBaseModel") + assert "revision" not in _params(function) + assert function.args.kwarg is not None, "**kwargs is what carries revision here" + + +@pytest.mark.parametrize( + "callee, minimum", + [("AutoConfig", 4), ("auto_processor", 2), ("_AutoTokenizer", 2)], +) +def test_vision_loads_forward_revision(callee, minimum): + function = _function(_tree(VISION), "from_pretrained", "FastBaseModel") + calls = [ + c for c in _calls(function, "from_pretrained") + if ast.unparse(c.func).startswith(callee) + ] + assert len(calls) >= minimum, f"expected >= {minimum} {callee} loads, found {len(calls)}" + for call in calls: + assert _revision_kwarg(call) is not None, f"{callee} at line {call.lineno} drops revision" + + +@pytest.mark.parametrize("name", ["load_correct_tokenizer", "_load_correct_tokenizer"]) +def test_tokenizer_helpers_accept_revision(name): + assert "revision" in _params(_function(_tree(TOKENIZER_UTILS), name)) + + +def test_tokenizer_helpers_forward_revision(): + tree = _tree(TOKENIZER_UTILS) + public = _function(tree, "load_correct_tokenizer") + inner = _calls(public, "_load_correct_tokenizer") + assert len(inner) == 1 and _revision_kwarg(inner[0]) is not None + + private = _function(tree, "_load_correct_tokenizer") + loads = _calls(private, "from_pretrained") + assert len(loads) >= 2, "expected the slow and fast tokenizer loads" + for call in loads: + assert _revision_kwarg(call) is not None, f"tokenizer load at line {call.lineno} drops revision" + + +def _load_gate(): + """Exec just _revision_for_resolved_repo, so no GPU-bound import is needed.""" + source = LOADER.read_text(encoding = "utf-8") + function = _function(ast.parse(source), "_revision_for_resolved_repo") + namespace = {"logger": types.SimpleNamespace(warning_once = lambda *a, **k: None)} + module = ast.Module(body = [function], type_ignores = []) + ast.fix_missing_locations(module) + exec(compile(module, str(LOADER), "exec"), namespace) + return namespace["_revision_for_resolved_repo"] + + +def test_revision_survives_when_the_repo_is_unchanged(): + # The reported case: a user's own repo is never in the mapper tables. + gate = _load_gate() + assert gate("my-branch", "myorg/my-ft", "myorg/my-ft") == "my-branch" + + +@pytest.mark.parametrize( + "model_name, old_model_name", + [ + ("unsloth/llama-3-8b-bnb-4bit", "meta-llama/Meta-Llama-3-8B"), # prequant mirror + ("unsloth/Qwen3-30B-A3B", "unsloth/Qwen3-30B-A3B-bnb-4bit"), # suffix strip + ("/tmp/unsloth-fp8-cache/model", "meta-llama/Meta-Llama-3-8B"), # fp8 temp dir + ], +) +def test_revision_is_dropped_once_the_repo_is_remapped(model_name, old_model_name): + # The ref only exists on the repo the caller named, so pinning it elsewhere + # would 404 or, worse, resolve a same-named branch on a different repo. + assert _load_gate()("abc123", model_name, old_model_name) is None + + +def test_no_revision_stays_none_even_when_remapped(): + gate = _load_gate() + assert gate(None, "unsloth/llama-3-8b-bnb-4bit", "meta-llama/Meta-Llama-3-8B") is None + + +def test_the_gate_warns_exactly_once_when_it_drops_a_revision(): + source = LOADER.read_text(encoding = "utf-8") + function = _function(ast.parse(source), "_revision_for_resolved_repo") + warnings = [] + namespace = {"logger": types.SimpleNamespace(warning_once = lambda m: warnings.append(m))} + module = ast.Module(body = [function], type_ignores = []) + ast.fix_missing_locations(module) + exec(compile(module, str(LOADER), "exec"), namespace) + + namespace["_revision_for_resolved_repo"]("abc123", "unsloth/x-bnb-4bit", "org/x") + assert len(warnings) == 1 + message = warnings[0] + # Both repos have to be named or the user cannot tell which load was silently redirected. + assert "abc123" in message and "org/x" in message and "unsloth/x-bnb-4bit" in message + assert "use_exact_model_name" in message + + +@pytest.mark.parametrize("function_name", ["from_pretrained"]) +def test_both_loader_paths_gate_the_revision_they_dispatch(function_name): + """Both public entry points must run the gate and pass its result on, while the + adapter load keeps the caller's original `revision` for old_model_name.""" + tree = _tree(LOADER) + for class_name in ("FastLanguageModel", "FastModel"): + function = _function(tree, function_name, class_name) + gate_calls = _calls(function, "_revision_for_resolved_repo") + assert len(gate_calls) == 1, f"{class_name} must gate revision exactly once" + assigned = [ + n for n in ast.walk(function) + if isinstance(n, ast.Assign) + and any(getattr(t, "id", None) == "base_revision" for t in n.targets) + ] + assert assigned, f"{class_name} must keep the gated value separate from revision" + used = [n for n in ast.walk(function) if isinstance(n, ast.Name) and n.id == "base_revision" + and isinstance(n.ctx, ast.Load)] + assert used, f"{class_name} computes base_revision but never dispatches it" diff --git a/unsloth/models/llama.py b/unsloth/models/llama.py index 4ae55d976b..62e41c53ab 100644 --- a/unsloth/models/llama.py +++ b/unsloth/models/llama.py @@ -2407,6 +2407,7 @@ def from_pretrained( model_name, token = token, attn_implementation = "sdpa", + revision = revision, ) _checkpoint_quant = getattr(_checkpoint_config, "quantization_config", None) if _checkpoint_quant is not None: @@ -2416,6 +2417,7 @@ def from_pretrained( model_name, token = token, attn_implementation = "sdpa", + revision = revision, ) model_config.model_name = model_name model_max_seq_length = model_config.max_position_embeddings @@ -2429,11 +2431,12 @@ def from_pretrained( preferred_attn_impl = resolve_attention_implementation(model_function, model_config) # Prefetch the repo (killable child) so the weight load is a cache hit. Runs after the - # AutoConfig/model-class check so an unsupported repo fails on its small config fetch. No - # revision: the load resolves model_name (maybe a remapped prequant repo) on its default branch. + # AutoConfig/model-class check so an unsupported repo fails on its small config fetch. + # Warm the same revision the load below uses, or the repo is downloaded twice. _prefetched = maybe_prefetch_hf_snapshot( model_name, token = token, + revision = revision, cache_dir = kwargs.get("cache_dir"), local_files_only = kwargs.get("local_files_only", False), # Skip the warm only for a real vLLM load; a num_labels classification load still goes @@ -2493,6 +2496,8 @@ def from_pretrained( cache_dir = _tokenizer_cache_dir, local_files_only = kwargs.get("local_files_only", False), tokenizer_only = True, + # Matches the tokenizer load below, which only pins its own repo. + revision = revision if _tokenizer_repo == model_name else None, ) has_rope_scaling = False @@ -2626,6 +2631,7 @@ def from_pretrained( token = token, trust_remote_code = trust_remote_code, attn_implementation = preferred_attn_impl, + revision = revision, **kwargs, ) # Defensive: ensure the task head is in a floating dtype, guarding @@ -2654,8 +2660,8 @@ def from_pretrained( model_name, local_files_only = kwargs.get("local_files_only", False), token = token, - # Weights load from the default branch (revision not forwarded), so read scales from there too. - revision = None, + # Read the scales from the same revision the weights came from. + revision = revision, subfolder = kwargs.get("subfolder"), cache_dir = kwargs.get("cache_dir"), variant = kwargs.get("variant"), @@ -2674,6 +2680,7 @@ def from_pretrained( token = token, trust_remote_code = trust_remote_code, attn_implementation = preferred_attn_impl, + revision = revision, **kwargs, ) else: @@ -2686,6 +2693,7 @@ def from_pretrained( max_position_embeddings = max_position_embeddings, trust_remote_code = trust_remote_code, attn_implementation = preferred_attn_impl, + revision = revision, **kwargs, ) # Attach dispatch hooks for bnb multi-device loads. @@ -2704,8 +2712,8 @@ def from_pretrained( model_name, local_files_only = kwargs.get("local_files_only", False), token = token, - # Weights load from the default branch (revision not forwarded), so read scales from there too. - revision = None, + # Read the scales from the same revision the weights came from. + revision = revision, subfolder = kwargs.get("subfolder"), cache_dir = kwargs.get("cache_dir"), variant = kwargs.get("variant"), @@ -2781,6 +2789,7 @@ def from_pretrained( token = token, trust_remote_code = trust_remote_code, fix_tokenizer = fix_tokenizer, + revision = revision if tokenizer_name == model_name else None, **_tokenizer_cache_kwargs, ) diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index ec979f811d..15d434ba8e 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -149,6 +149,23 @@ def _strip_unsloth_bnb_4bit_suffix(model_name: str) -> str: return s +def _revision_for_resolved_repo(revision, model_name, old_model_name): + """Drop `revision` once the requested repo has been remapped to another one. + + A revision names a branch/tag/SHA on the repo the caller asked for, but from_pretrained + may resolve model_name to a different repo (a pre-quantized mirror, an fp8 temp dir, a + ModelScope snapshot, a -bnb-4bit strip), where that ref does not exist. + """ + if revision is None or model_name == old_model_name: + return revision + logger.warning_once( + f"Unsloth: Ignoring revision = `{revision}` since `{old_model_name}` resolved to " + f"`{model_name}`, which does not have that revision. " + "Pass `use_exact_model_name = True` to load your repo as-is." + ) + return None + + def _config_get( config, field_name, @@ -850,6 +867,10 @@ def from_pretrained( if fast_inference: fast_inference, model_name = fast_inference_setup(model_name, model_config) + # Last point model_name can change. Kept separate from `revision`, which still + # belongs to old_model_name for the adapter load further below. + base_revision = _revision_for_resolved_repo(revision, model_name, old_model_name) + load_in_4bit_kwargs = load_in_4bit load_in_8bit_kwargs = load_in_8bit if quantization_config is not None and not fast_inference: @@ -876,7 +897,7 @@ def from_pretrained( model_patcher = dispatch_model, tokenizer_name = tokenizer_name, trust_remote_code = trust_remote_code, - revision = revision if not is_peft else None, + revision = base_revision if not is_peft else None, fast_inference = fast_inference, gpu_memory_utilization = gpu_memory_utilization, float8_kv_cache = float8_kv_cache, @@ -1798,6 +1819,10 @@ def _dispatch_diffusion(): load_in_4bit_kwargs = False load_in_8bit_kwargs = False + # Kept separate from `revision`, which still belongs to old_model_name for the + # adapter load further below. FastBaseModel remaps again via fast_inference_setup. + base_revision = _revision_for_resolved_repo(revision, model_name, old_model_name) + model, tokenizer = FastBaseModel.from_pretrained( model_name = model_name, max_seq_length = max_seq_length, @@ -1809,7 +1834,7 @@ def _dispatch_diffusion(): token = token, device_map = device_map, trust_remote_code = trust_remote_code, - revision = revision if not is_peft else None, + revision = base_revision if not is_peft else None, model_types = model_types, tokenizer_name = tokenizer_name, auto_model = auto_model, diff --git a/unsloth/models/loader_utils.py b/unsloth/models/loader_utils.py index e8d0fa657c..debf2f98ba 100644 --- a/unsloth/models/loader_utils.py +++ b/unsloth/models/loader_utils.py @@ -1318,6 +1318,7 @@ def _load_pretrained_tokenizer_fast( trust_remote_code = False, cache_dir = None, local_files_only = False, + revision = None, ): """Load ``PreTrainedTokenizerFast`` without Hub metadata probes when cached/offline. @@ -1344,6 +1345,7 @@ def _load_pretrained_tokenizer_fast( trust_remote_code = trust_remote_code, cache_dir = cache_dir, local_files_only = lfo, + revision = revision, ) diff --git a/unsloth/models/vision.py b/unsloth/models/vision.py index 13cd582437..fe6057650e 100644 --- a/unsloth/models/vision.py +++ b/unsloth/models/vision.py @@ -664,6 +664,7 @@ def _construct_vlm_processor_fallback( trust_remote_code, cache_dir = None, local_files_only = False, + revision = None, ): """Build a VLM processor manually when AutoProcessor.from_pretrained fails (some VLMs have unresolvable tokenizer_class entries): load the image processor + tokenizer @@ -688,6 +689,7 @@ def _construct_vlm_processor_fallback( trust_remote_code = trust_remote_code, cache_dir = cache_dir, local_files_only = local_files_only, + revision = revision, ) # Load tokenizer via PreTrainedTokenizerFast (bypasses tokenizer_class check). # Resolve the cached snapshot first so transformers does not call model_info (#7481). @@ -698,6 +700,7 @@ def _construct_vlm_processor_fallback( trust_remote_code = trust_remote_code, cache_dir = cache_dir, local_files_only = local_files_only, + revision = revision, ) # Read tokenizer_config.json for special tokens: prefer the local file (offline # / local checkpoint dir), else hf_hub_download with local_files_only forwarded. @@ -723,6 +726,7 @@ def _construct_vlm_processor_fallback( "tokenizer_config.json", token = token, cache_dir = cache_dir, + revision = revision, local_files_only = local_files_only, ) with open(config_path, "r", encoding = "utf-8") as f: @@ -756,6 +760,7 @@ def _construct_vlm_processor_fallback( trust_remote_code = trust_remote_code, cache_dir = cache_dir, local_files_only = local_files_only, + revision = revision, ) proc_class_name = PROCESSOR_MAPPING_NAMES.get(config.model_type) except Exception as _e: @@ -862,6 +867,12 @@ def from_pretrained( if os.environ.get("UNSLOTH_MODEL_NAME", "") == "": os.environ["UNSLOTH_MODEL_NAME"] = model_name.lower() + # Read revision out of kwargs rather than binding it in the signature: the weight + # load below forwards **kwargs, so a named parameter would drop it from there. + # It names a ref on the repo as passed in, so pin that before any remap. + _revision = kwargs.get("revision") + _revision_repo = model_name + # Resolve text-only before the is_vlm / vLLM checks so is_vlm stays consistent; # skip the vision tower only for families with their own text decoder (Gemma 3). #5816 if text_only and auto_config is None: @@ -870,6 +881,7 @@ def from_pretrained( token = token, trust_remote_code = trust_remote_code, local_files_only = local_files_only, + revision = _revision, ) if text_only and hasattr(auto_config, "vision_config"): parent_config = auto_config @@ -1027,6 +1039,7 @@ def from_pretrained( token = token, trust_remote_code = trust_remote_code, local_files_only = local_files_only, + revision = _revision, ) model_class = resolve_model_class(auto_model, auto_config) attn_impl = resolve_attention_implementation( @@ -1108,6 +1121,7 @@ def from_pretrained( and _tokenizer_repo != model_name ) if _warm_tokenizer_repo: + # No revision: this only runs for a repo other than the one it belongs to. maybe_prefetch_hf_snapshot( _tokenizer_repo, token = token, @@ -1182,6 +1196,7 @@ def from_pretrained( token = token, trust_remote_code = trust_remote_code, local_files_only = local_files_only, + revision = _revision, ) if hasattr(auto_config, "quantization_config"): from transformers.quantizers.auto import ( @@ -1236,6 +1251,7 @@ def from_pretrained( token = token, trust_remote_code = trust_remote_code, local_files_only = local_files_only, + revision = _revision, ) _set_attn_impl(auto_config, config_attn_impl) model_config = auto_config @@ -1433,6 +1449,9 @@ def from_pretrained( # Counteract saved tokenizers tokenizer_name = model_name if tokenizer_name is None else tokenizer_name + # The tokenizer repo can differ from the one the revision belongs to (a caller + # override, or model_name remapped by fast_inference_setup above). + _tokenizer_revision = _revision if tokenizer_name == _revision_repo else None # On the vLLM path the tokenizer warm was deferred (fast_inference_setup may remap model_name). # Warm the now-final tokenizer repo so the load below hits the cache (a cached/local repo is a no-op). @@ -1440,7 +1459,8 @@ def from_pretrained( maybe_prefetch_hf_snapshot( tokenizer_name, token = token, - revision = kwargs.get("revision"), + # Match the tokenizer load below, which only pins its own repo. + revision = _tokenizer_revision, cache_dir = kwargs.get("cache_dir"), local_files_only = kwargs.get("local_files_only", False), tokenizer_only = True, @@ -1485,6 +1505,7 @@ def _acquire_processor(lfo): trust_remote_code = trust_remote_code, cache_dir = kwargs.get("cache_dir"), local_files_only = lfo, + revision = _tokenizer_revision, ) except Exception as _e: _tok = None @@ -1498,6 +1519,7 @@ def _acquire_processor(lfo): trust_remote_code = trust_remote_code, cache_dir = kwargs.get("cache_dir"), local_files_only = lfo, + revision = _tokenizer_revision, ) except Exception as _e: _err = _e @@ -1528,6 +1550,7 @@ def _acquire_processor(lfo): trust_remote_code, cache_dir = kwargs.get("cache_dir"), local_files_only = lfo, + revision = _tokenizer_revision, ) except Exception as _fe: _fallback, _fb_err = None, _fe @@ -1614,6 +1637,7 @@ def _is_degraded_vlm(_t): trust_remote_code = trust_remote_code, cache_dir = kwargs.get("cache_dir"), local_files_only = local_files_only, + revision = _tokenizer_revision, ) model, _fallback_tok = patch_tokenizer(model, _fallback_tok) # Re-attach as processor wrapper if original was a processor @@ -1650,6 +1674,7 @@ def _last_resort_tokenizer(lfo): trust_remote_code = trust_remote_code, cache_dir = kwargs.get("cache_dir"), local_files_only = lfo, + revision = _tokenizer_revision, ) except Exception: return _load_pretrained_tokenizer_fast( @@ -1659,6 +1684,7 @@ def _last_resort_tokenizer(lfo): trust_remote_code = trust_remote_code, cache_dir = kwargs.get("cache_dir"), local_files_only = lfo, + revision = _tokenizer_revision, ) _last_resort_err = None diff --git a/unsloth/tokenizer_utils.py b/unsloth/tokenizer_utils.py index c7f61288d5..46ec60c37c 100644 --- a/unsloth/tokenizer_utils.py +++ b/unsloth/tokenizer_utils.py @@ -568,6 +568,7 @@ def _load_correct_tokenizer( trust_remote_code = False, cache_dir = "huggingface_tokenizers_cache", fix_tokenizer = True, + revision = None, ): if IS_COLAB_ENVIRONMENT: cache_dir = cache_dir @@ -596,6 +597,7 @@ def _load_correct_tokenizer( legacy = False, from_slow = True, cache_dir = cache_dir, + revision = revision, ) except: slow_tokenizer = None @@ -614,6 +616,7 @@ def _load_correct_tokenizer( token = token, trust_remote_code = trust_remote_code, cache_dir = cache_dir, + revision = revision, ) if not fix_tokenizer or tokenizer_name.lower() in IGNORED_TOKENIZER_NAMES: @@ -669,6 +672,7 @@ def load_correct_tokenizer( trust_remote_code = False, cache_dir = "huggingface_tokenizers_cache", fix_tokenizer = True, + revision = None, ): tokenizer = _load_correct_tokenizer( tokenizer_name = tokenizer_name, @@ -678,6 +682,7 @@ def load_correct_tokenizer( trust_remote_code = trust_remote_code, cache_dir = cache_dir, fix_tokenizer = fix_tokenizer, + revision = revision, ) if fix_tokenizer: From 24b5e964efe681d0743d43dc567360466f0048bf Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Sun, 2 Aug 2026 14:51:56 +0000 Subject: [PATCH 05/20] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- tests/python/test_revision_forwarding.py | 49 ++++++++++++++++-------- 1 file changed, 33 insertions(+), 16 deletions(-) diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py index fcaffd11f9..62acd05e94 100644 --- a/tests/python/test_revision_forwarding.py +++ b/tests/python/test_revision_forwarding.py @@ -26,7 +26,11 @@ def _tree(path): return ast.parse(path.read_text(encoding = "utf-8")) -def _function(tree, name, class_name = None): +def _function( + tree, + name, + class_name = None, +): body = tree.body if class_name is not None: classes = [n for n in body if isinstance(n, ast.ClassDef) and n.name == class_name] @@ -45,7 +49,8 @@ def _params(function): def _calls(function, callee): """Every Call whose dotted name ends with `callee`.""" return [ - node for node in ast.walk(function) + node + for node in ast.walk(function) if isinstance(node, ast.Call) and ast.unparse(node.func).split(".")[-1] == callee ] @@ -61,24 +66,30 @@ def test_fast_llama_model_reads_its_revision_argument(): """The whole of #3544: the parameter existed but had zero reads.""" function = _function(_tree(LLAMA), "from_pretrained", "FastLlamaModel") assert "revision" in _params(function) - loads = [n for n in ast.walk(function) if isinstance(n, ast.Name) and n.id == "revision" - and isinstance(n.ctx, ast.Load)] + loads = [ + n + for n in ast.walk(function) + if isinstance(n, ast.Name) and n.id == "revision" and isinstance(n.ctx, ast.Load) + ] assert loads, "revision is accepted but never read" @pytest.mark.parametrize( "callee, minimum", [ - ("AutoConfig", 2), # checkpoint probe + main config - ("AutoModelForCausalLM", 2), # user-config and plain branches + ("AutoConfig", 2), # checkpoint probe + main config + ("AutoModelForCausalLM", 2), # user-config and plain branches ("AutoModelForSequenceClassification", 1), ("load_correct_tokenizer", 1), ], ) def test_llama_loads_forward_revision(callee, minimum): function = _function(_tree(LLAMA), "from_pretrained", "FastLlamaModel") - calls = _calls(function, "from_pretrained") if callee != "load_correct_tokenizer" else \ - _calls(function, "load_correct_tokenizer") + calls = ( + _calls(function, "from_pretrained") + if callee != "load_correct_tokenizer" + else _calls(function, "load_correct_tokenizer") + ) if callee != "load_correct_tokenizer": calls = [c for c in calls if ast.unparse(c.func).startswith(callee)] assert len(calls) >= minimum, f"expected >= {minimum} {callee} loads, found {len(calls)}" @@ -91,7 +102,8 @@ def test_llama_does_not_pass_revision_to_load_vllm(): so putting one in that dict is an unconditional TypeError on the vLLM path.""" function = _function(_tree(LLAMA), "from_pretrained", "FastLlamaModel") dicts = [ - node.value for node in ast.walk(function) + node.value + for node in ast.walk(function) if isinstance(node, ast.Assign) and any(getattr(t, "id", None) == "load_vllm_kwargs" for t in node.targets) and isinstance(node.value, ast.Call) @@ -116,8 +128,7 @@ def test_fast_base_model_does_not_bind_revision(): def test_vision_loads_forward_revision(callee, minimum): function = _function(_tree(VISION), "from_pretrained", "FastBaseModel") calls = [ - c for c in _calls(function, "from_pretrained") - if ast.unparse(c.func).startswith(callee) + c for c in _calls(function, "from_pretrained") if ast.unparse(c.func).startswith(callee) ] assert len(calls) >= minimum, f"expected >= {minimum} {callee} loads, found {len(calls)}" for call in calls: @@ -139,7 +150,9 @@ def test_tokenizer_helpers_forward_revision(): loads = _calls(private, "from_pretrained") assert len(loads) >= 2, "expected the slow and fast tokenizer loads" for call in loads: - assert _revision_kwarg(call) is not None, f"tokenizer load at line {call.lineno} drops revision" + assert ( + _revision_kwarg(call) is not None + ), f"tokenizer load at line {call.lineno} drops revision" def _load_gate(): @@ -163,7 +176,7 @@ def test_revision_survives_when_the_repo_is_unchanged(): "model_name, old_model_name", [ ("unsloth/llama-3-8b-bnb-4bit", "meta-llama/Meta-Llama-3-8B"), # prequant mirror - ("unsloth/Qwen3-30B-A3B", "unsloth/Qwen3-30B-A3B-bnb-4bit"), # suffix strip + ("unsloth/Qwen3-30B-A3B", "unsloth/Qwen3-30B-A3B-bnb-4bit"), # suffix strip ("/tmp/unsloth-fp8-cache/model", "meta-llama/Meta-Llama-3-8B"), # fp8 temp dir ], ) @@ -205,11 +218,15 @@ def test_both_loader_paths_gate_the_revision_they_dispatch(function_name): gate_calls = _calls(function, "_revision_for_resolved_repo") assert len(gate_calls) == 1, f"{class_name} must gate revision exactly once" assigned = [ - n for n in ast.walk(function) + n + for n in ast.walk(function) if isinstance(n, ast.Assign) and any(getattr(t, "id", None) == "base_revision" for t in n.targets) ] assert assigned, f"{class_name} must keep the gated value separate from revision" - used = [n for n in ast.walk(function) if isinstance(n, ast.Name) and n.id == "base_revision" - and isinstance(n.ctx, ast.Load)] + used = [ + n + for n in ast.walk(function) + if isinstance(n, ast.Name) and n.id == "base_revision" and isinstance(n.ctx, ast.Load) + ] assert used, f"{class_name} computes base_revision but never dispatches it" From 8e36e72a9393ce882f6d1e9de133b5f43198488c Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 2 Aug 2026 14:58:17 +0000 Subject: [PATCH 06/20] Tighten the revision comments --- unsloth/models/llama.py | 6 +++--- unsloth/models/loader.py | 8 ++++---- unsloth/models/vision.py | 11 +++++------ 3 files changed, 12 insertions(+), 13 deletions(-) diff --git a/unsloth/models/llama.py b/unsloth/models/llama.py index 62e41c53ab..5cc34aa218 100644 --- a/unsloth/models/llama.py +++ b/unsloth/models/llama.py @@ -2432,7 +2432,7 @@ def from_pretrained( # Prefetch the repo (killable child) so the weight load is a cache hit. Runs after the # AutoConfig/model-class check so an unsupported repo fails on its small config fetch. - # Warm the same revision the load below uses, or the repo is downloaded twice. + # Warm the same revision the load uses, or the repo downloads twice. _prefetched = maybe_prefetch_hf_snapshot( model_name, token = token, @@ -2660,7 +2660,7 @@ def from_pretrained( model_name, local_files_only = kwargs.get("local_files_only", False), token = token, - # Read the scales from the same revision the weights came from. + # Read scales from the same revision as the weights. revision = revision, subfolder = kwargs.get("subfolder"), cache_dir = kwargs.get("cache_dir"), @@ -2712,7 +2712,7 @@ def from_pretrained( model_name, local_files_only = kwargs.get("local_files_only", False), token = token, - # Read the scales from the same revision the weights came from. + # Read scales from the same revision as the weights. revision = revision, subfolder = kwargs.get("subfolder"), cache_dir = kwargs.get("cache_dir"), diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index 15d434ba8e..96481c9040 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -867,8 +867,8 @@ def from_pretrained( if fast_inference: fast_inference, model_name = fast_inference_setup(model_name, model_config) - # Last point model_name can change. Kept separate from `revision`, which still - # belongs to old_model_name for the adapter load further below. + # Last point model_name can change. Separate from `revision`, which the adapter + # load below still needs for old_model_name. base_revision = _revision_for_resolved_repo(revision, model_name, old_model_name) load_in_4bit_kwargs = load_in_4bit @@ -1819,8 +1819,8 @@ def _dispatch_diffusion(): load_in_4bit_kwargs = False load_in_8bit_kwargs = False - # Kept separate from `revision`, which still belongs to old_model_name for the - # adapter load further below. FastBaseModel remaps again via fast_inference_setup. + # Separate from `revision`, which the adapter load below still needs for + # old_model_name. FastBaseModel remaps again via fast_inference_setup. base_revision = _revision_for_resolved_repo(revision, model_name, old_model_name) model, tokenizer = FastBaseModel.from_pretrained( diff --git a/unsloth/models/vision.py b/unsloth/models/vision.py index fe6057650e..5ea436f90f 100644 --- a/unsloth/models/vision.py +++ b/unsloth/models/vision.py @@ -867,9 +867,8 @@ def from_pretrained( if os.environ.get("UNSLOTH_MODEL_NAME", "") == "": os.environ["UNSLOTH_MODEL_NAME"] = model_name.lower() - # Read revision out of kwargs rather than binding it in the signature: the weight - # load below forwards **kwargs, so a named parameter would drop it from there. - # It names a ref on the repo as passed in, so pin that before any remap. + # Read revision from kwargs, not the signature: the weight load below forwards + # **kwargs, so binding it would drop it there. Pin its repo before any remap. _revision = kwargs.get("revision") _revision_repo = model_name @@ -1121,7 +1120,7 @@ def from_pretrained( and _tokenizer_repo != model_name ) if _warm_tokenizer_repo: - # No revision: this only runs for a repo other than the one it belongs to. + # No revision: this only runs when the repo differs from the revision's repo. maybe_prefetch_hf_snapshot( _tokenizer_repo, token = token, @@ -1449,8 +1448,8 @@ def from_pretrained( # Counteract saved tokenizers tokenizer_name = model_name if tokenizer_name is None else tokenizer_name - # The tokenizer repo can differ from the one the revision belongs to (a caller - # override, or model_name remapped by fast_inference_setup above). + # The tokenizer repo can differ from the revision's repo (a caller override, or + # model_name remapped by fast_inference_setup above). _tokenizer_revision = _revision if tokenizer_name == _revision_repo else None # On the vLLM path the tokenizer warm was deferred (fast_inference_setup may remap model_name). From 2c2b66d1502fbc88161d7f2618f39552a7ab8b91 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 2 Aug 2026 15:22:49 +0000 Subject: [PATCH 07/20] Gate the revision before the config probes, and never mix refs Four fixes from review: - The gate ran after the AutoConfig and PeftConfig probes, which already used the raw revision against the resolved name, so a pinned load_in_4bit load failed against the mirror instead of warning. Gate right after the resolution block and point both probes at the gated value, then re-gate before dispatch for the later fast_inference_setup remap. Feeding the second call the first result keeps the warning to one. - On a PEFT load model_name is necessarily the base model, so the late gate warned "Ignoring revision" for every versioned adapter and told the caller to pass use_exact_model_name, which cannot stop an adapter resolving its base. Skip the late gate for PEFT; PeftModel.from_pretrained already loads the adapter with the caller's revision. - load_vllm takes no revision, so vLLM fetches the default branch. Pinning only the config and the tokenizer put two refs in one model, which is worse than the old behaviour of ignoring the revision outright. Drop the pin with a warning before the config load whenever vLLM owns the weights. - _hub_repo_or_local_path resolved a cached snapshot without the revision, so an offline or local_files_only tokenizer load silently got the default ref: a revision handed to from_pretrained cannot re-point a local directory. Thread it into _resolve_hub_repo_local_dir and both call sites. Five new tests, one per fix, all failing before it. --- tests/python/test_revision_forwarding.py | 116 +++++++++++++++++++---- unsloth/models/llama.py | 8 ++ unsloth/models/loader.py | 32 ++++--- unsloth/models/loader_utils.py | 5 + unsloth/models/vision.py | 11 +++ 5 files changed, 143 insertions(+), 29 deletions(-) diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py index 62acd05e94..dc9a861b7d 100644 --- a/tests/python/test_revision_forwarding.py +++ b/tests/python/test_revision_forwarding.py @@ -20,6 +20,7 @@ LOADER = REPO / "unsloth" / "models" / "loader.py" VISION = REPO / "unsloth" / "models" / "vision.py" TOKENIZER_UTILS = REPO / "unsloth" / "tokenizer_utils.py" +LOADER_UTILS = REPO / "unsloth" / "models" / "loader_utils.py" def _tree(path): @@ -208,25 +209,104 @@ def test_the_gate_warns_exactly_once_when_it_drops_a_revision(): assert "use_exact_model_name" in message -@pytest.mark.parametrize("function_name", ["from_pretrained"]) -def test_both_loader_paths_gate_the_revision_they_dispatch(function_name): - """Both public entry points must run the gate and pass its result on, while the - adapter load keeps the caller's original `revision` for old_model_name.""" +def test_both_loader_paths_gate_before_and_after_resolution(): + """The gate has to run before the AutoConfig / PeftConfig probes, or a pinned 4bit + load fails against the mirror instead of warning, and again after the last remap.""" tree = _tree(LOADER) for class_name in ("FastLanguageModel", "FastModel"): - function = _function(tree, function_name, class_name) - gate_calls = _calls(function, "_revision_for_resolved_repo") - assert len(gate_calls) == 1, f"{class_name} must gate revision exactly once" - assigned = [ - n - for n in ast.walk(function) - if isinstance(n, ast.Assign) - and any(getattr(t, "id", None) == "base_revision" for t in n.targets) + function = _function(tree, "from_pretrained", class_name) + gates = _calls(function, "_revision_for_resolved_repo") + assert len(gates) == 2, f"{class_name} needs an early and a late gate, found {len(gates)}" + early, late = sorted(gates, key = lambda c: c.lineno) + + probes = [ + c for c in _calls(function, "from_pretrained") + if ast.unparse(c.func).split(".")[0] in ("AutoConfig", "PeftConfig") ] - assert assigned, f"{class_name} must keep the gated value separate from revision" - used = [ - n - for n in ast.walk(function) - if isinstance(n, ast.Name) and n.id == "base_revision" and isinstance(n.ctx, ast.Load) + assert probes, f"{class_name} has no config probe" + gated = 0 + for probe in probes: + assert probe.lineno > early.lineno, "the gate must precede the config probes" + keyword = _revision_kwarg(probe) + if keyword is None: + continue # the PEFT base-model probe deliberately pins nothing + assert getattr(keyword.value, "id", None) == "base_revision", ( + f"probe at line {probe.lineno} uses the ungated revision" + ) + gated += 1 + assert gated >= 2, f"{class_name} must gate its AutoConfig and PeftConfig probes" + + # The late gate feeds on base_revision so an already-dropped one warns only once. + assert getattr(late.args[0], "id", None) == "base_revision" + + +def test_the_late_gate_is_skipped_for_peft(): + """On a PEFT load model_name is necessarily the base model, so the remap warning + would fire for every versioned adapter while PeftModel loads the ref correctly.""" + tree = _tree(LOADER) + for class_name in ("FastLanguageModel", "FastModel"): + function = _function(tree, "from_pretrained", class_name) + late = sorted(_calls(function, "_revision_for_resolved_repo"), key = lambda c: c.lineno)[-1] + guards = [ + n for n in ast.walk(function) + if isinstance(n, ast.If) + and ast.unparse(n.test).replace(" ", "") == "notis_peft" + and n.lineno <= late.lineno <= n.end_lineno ] - assert used, f"{class_name} computes base_revision but never dispatches it" + assert guards, f"{class_name}'s late gate must sit under `if not is_peft`" + + +def test_the_adapter_load_keeps_the_callers_revision(): + """`revision` names the adapter repo, so PeftModel must get the ungated value.""" + tree = _tree(LOADER) + for class_name in ("FastLanguageModel", "FastModel"): + function = _function(tree, "from_pretrained", class_name) + peft_loads = [ + c for c in _calls(function, "from_pretrained") + if ast.unparse(c.func).startswith("PeftModel") + ] + assert peft_loads, f"{class_name} has no PeftModel load" + for call in peft_loads: + keyword = _revision_kwarg(call) + assert keyword is not None and getattr(keyword.value, "id", None) == "revision" + + +@pytest.mark.parametrize("path, flag", [(LLAMA, "revision"), (VISION, "_revision")]) +def test_a_pinned_load_does_not_mix_refs_with_vllm(path, flag): + """load_vllm takes no revision, so vLLM fetches the default branch. Pinning only the + config and tokenizer would put two refs in one model, so the pin is dropped instead.""" + source = path.read_text(encoding = "utf-8") + tree = ast.parse(source) + name = "FastLlamaModel" if path is LLAMA else "FastBaseModel" + function = _function(tree, "from_pretrained", name) + clears = [ + n for n in ast.walk(function) + if isinstance(n, ast.Assign) + and any(getattr(t, "id", None) == flag for t in n.targets) + and isinstance(n.value, ast.Constant) and n.value.value is None + ] + assert clears, f"{path.name} never drops the revision on the vLLM path" + # It must happen before the config load, or the config is pinned and the weights are not. + configs = [ + c for c in _calls(function, "from_pretrained") + if ast.unparse(c.func).startswith("AutoConfig") + ] + assert configs + assert min(c.lineno for c in clears) < min(c.lineno for c in configs) + + +def test_local_snapshot_resolution_takes_the_revision(): + """A local snapshot dir cannot be re-pointed by a revision handed to from_pretrained, + so the cache resolution itself has to select the requested ref.""" + tree = _tree(LOADER_UTILS) + for name in ("_resolve_hub_repo_local_dir", "_hub_repo_or_local_path"): + function = _function(tree, name) + assert "revision" in _params(function), f"{name} must accept revision" + resolver = _function(tree, "_resolve_hub_repo_local_dir") + downloads = _calls(resolver, "hf_hub_download") + assert downloads, "expected the cache probe download" + for call in downloads: + assert _revision_kwarg(call) is not None + wrapper = _function(tree, "_hub_repo_or_local_path") + inner = _calls(wrapper, "_resolve_hub_repo_local_dir") + assert inner and all(_revision_kwarg(c) is not None for c in inner) diff --git a/unsloth/models/llama.py b/unsloth/models/llama.py index 5cc34aa218..3de5b5f8c5 100644 --- a/unsloth/models/llama.py +++ b/unsloth/models/llama.py @@ -2338,6 +2338,14 @@ def from_pretrained( raise RuntimeError( "Unsloth: `unsloth_vllm_standby` is True, but environment variable `UNSLOTH_VLLM_STANDBY` is not set to 1!" ) + if revision is not None: + # load_vllm takes no revision, so vLLM fetches the default branch. Pinning + # only the config and tokenizer would mix two refs in one model. + logger.warning_once( + f"Unsloth: Ignoring revision = `{revision}` since vLLM loads weights from " + "the default branch. Use `fast_inference = False` to load a pinned revision." + ) + revision = None token = hf_login(token) if model_patcher is None: diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index 96481c9040..ce2c335ec7 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -574,6 +574,10 @@ def from_pretrained( from modelscope import snapshot_download model_name = snapshot_download(model_name) + # Gate before the probe below, or a pinned 4bit load fails against the mirror + # instead of warning. Kept separate: `revision` still names the adapter repo. + base_revision = _revision_for_resolved_repo(revision, model_name, old_model_name) + # First check if it's a normal model via AutoConfig from huggingface_hub.utils import ( disable_progress_bars, @@ -596,7 +600,7 @@ def from_pretrained( model_config = AutoConfig.from_pretrained( model_name, token = token, - revision = revision, + revision = base_revision, trust_remote_code = trust_remote_code, local_files_only = local_files_only, ) @@ -623,7 +627,7 @@ def from_pretrained( peft_config = PeftConfig.from_pretrained( model_name, token = token, - revision = revision, + revision = base_revision, trust_remote_code = trust_remote_code, local_files_only = local_files_only, ) @@ -867,9 +871,10 @@ def from_pretrained( if fast_inference: fast_inference, model_name = fast_inference_setup(model_name, model_config) - # Last point model_name can change. Separate from `revision`, which the adapter - # load below still needs for old_model_name. - base_revision = _revision_for_resolved_repo(revision, model_name, old_model_name) + # model_name can move once more here. Skip for PEFT: model_name is then the base + # model, and `revision` names the adapter that PeftModel.from_pretrained loads below. + if not is_peft: + base_revision = _revision_for_resolved_repo(base_revision, model_name, old_model_name) load_in_4bit_kwargs = load_in_4bit load_in_8bit_kwargs = load_in_8bit @@ -1297,6 +1302,10 @@ def from_pretrained( from modelscope import snapshot_download model_name = snapshot_download(model_name) + # Gate before the probe below, or a pinned 4bit load fails against the mirror + # instead of warning. Kept separate: `revision` still names the adapter repo. + base_revision = _revision_for_resolved_repo(revision, model_name, old_model_name) + # First check if it's a normal model via AutoConfig from huggingface_hub.utils import ( disable_progress_bars, @@ -1330,7 +1339,7 @@ def _dispatch_diffusion(): token = token, device_map = device_map, trust_remote_code = trust_remote_code, - revision = revision, + revision = base_revision, **kwargs, ) @@ -1340,7 +1349,7 @@ def _dispatch_diffusion(): model_config = AutoConfig.from_pretrained( model_name, token = token, - revision = revision, + revision = base_revision, trust_remote_code = trust_remote_code, local_files_only = local_files_only, ) @@ -1373,7 +1382,7 @@ def _dispatch_diffusion(): peft_config = PeftConfig.from_pretrained( model_name, token = token, - revision = revision, + revision = base_revision, trust_remote_code = trust_remote_code, local_files_only = local_files_only, ) @@ -1819,9 +1828,10 @@ def _dispatch_diffusion(): load_in_4bit_kwargs = False load_in_8bit_kwargs = False - # Separate from `revision`, which the adapter load below still needs for - # old_model_name. FastBaseModel remaps again via fast_inference_setup. - base_revision = _revision_for_resolved_repo(revision, model_name, old_model_name) + # FastBaseModel remaps again via fast_inference_setup. Skip for PEFT: model_name is + # then the base model, and `revision` names the adapter PeftModel loads below. + if not is_peft: + base_revision = _revision_for_resolved_repo(base_revision, model_name, old_model_name) model, tokenizer = FastBaseModel.from_pretrained( model_name = model_name, diff --git a/unsloth/models/loader_utils.py b/unsloth/models/loader_utils.py index debf2f98ba..68d55dc404 100644 --- a/unsloth/models/loader_utils.py +++ b/unsloth/models/loader_utils.py @@ -1214,6 +1214,7 @@ def _resolve_hub_repo_local_dir( *, token = None, cache_dir = None, + revision = None, # Default closed: a "resolve local dir" helper must not download. False here # means five filenames each retried with backoff before it gives up. local_files_only = True, @@ -1249,6 +1250,7 @@ def _resolve_hub_repo_local_dir( token = token, cache_dir = cache_dir, local_files_only = local_files_only, + revision = revision, ) if path and os.path.isfile(path): return os.path.dirname(path) @@ -1286,6 +1288,7 @@ def _hub_repo_or_local_path( cache_dir = None, local_files_only = False, filenames = None, + revision = None, ): """Prefer a cached snapshot path over a Hub repo id when offline or ``local_files_only``.""" if isinstance(repo_id, str) and os.path.isdir(repo_id): @@ -1298,6 +1301,7 @@ def _hub_repo_or_local_path( token = token, cache_dir = cache_dir, local_files_only = True, + revision = revision, filenames = filenames or ( "tokenizer_config.json", @@ -1332,6 +1336,7 @@ def _load_pretrained_tokenizer_fast( token = token, cache_dir = cache_dir, local_files_only = lfo, + revision = revision, filenames = ( "tokenizer_config.json", "tokenizer.json", diff --git a/unsloth/models/vision.py b/unsloth/models/vision.py index 5ea436f90f..cab28d9794 100644 --- a/unsloth/models/vision.py +++ b/unsloth/models/vision.py @@ -681,6 +681,7 @@ def _construct_vlm_processor_fallback( token = token, cache_dir = cache_dir, local_files_only = local_files_only, + revision = revision, ) # Load image processor image_processor = AutoImageProcessor.from_pretrained( @@ -870,6 +871,15 @@ def from_pretrained( # Read revision from kwargs, not the signature: the weight load below forwards # **kwargs, so binding it would drop it there. Pin its repo before any remap. _revision = kwargs.get("revision") + if _revision is not None and fast_inference and is_vLLM_available(): + # load_vllm takes no revision, so vLLM fetches the default branch. Pinning only + # the config and tokenizer would mix two refs in one model. + logger.warning_once( + f"Unsloth: Ignoring revision = `{_revision}` since vLLM loads weights from " + "the default branch. Use `fast_inference = False` to load a pinned revision." + ) + _revision = None + kwargs.pop("revision", None) _revision_repo = model_name # Resolve text-only before the is_vlm / vLLM checks so is_vlm stays consistent; @@ -1664,6 +1674,7 @@ def _last_resort_tokenizer(lfo): token = token, cache_dir = kwargs.get("cache_dir"), local_files_only = lfo, + revision = _tokenizer_revision, ) try: return _AutoTokenizer.from_pretrained( From 0e9f7eaa8ab06a3d6c667ddec4764f5d05ac6699 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Sun, 2 Aug 2026 15:23:38 +0000 Subject: [PATCH 08/20] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- tests/python/test_revision_forwarding.py | 24 +++++++++++++++--------- 1 file changed, 15 insertions(+), 9 deletions(-) diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py index dc9a861b7d..7e0ec45a7e 100644 --- a/tests/python/test_revision_forwarding.py +++ b/tests/python/test_revision_forwarding.py @@ -220,7 +220,8 @@ def test_both_loader_paths_gate_before_and_after_resolution(): early, late = sorted(gates, key = lambda c: c.lineno) probes = [ - c for c in _calls(function, "from_pretrained") + c + for c in _calls(function, "from_pretrained") if ast.unparse(c.func).split(".")[0] in ("AutoConfig", "PeftConfig") ] assert probes, f"{class_name} has no config probe" @@ -230,9 +231,9 @@ def test_both_loader_paths_gate_before_and_after_resolution(): keyword = _revision_kwarg(probe) if keyword is None: continue # the PEFT base-model probe deliberately pins nothing - assert getattr(keyword.value, "id", None) == "base_revision", ( - f"probe at line {probe.lineno} uses the ungated revision" - ) + assert ( + getattr(keyword.value, "id", None) == "base_revision" + ), f"probe at line {probe.lineno} uses the ungated revision" gated += 1 assert gated >= 2, f"{class_name} must gate its AutoConfig and PeftConfig probes" @@ -248,7 +249,8 @@ def test_the_late_gate_is_skipped_for_peft(): function = _function(tree, "from_pretrained", class_name) late = sorted(_calls(function, "_revision_for_resolved_repo"), key = lambda c: c.lineno)[-1] guards = [ - n for n in ast.walk(function) + n + for n in ast.walk(function) if isinstance(n, ast.If) and ast.unparse(n.test).replace(" ", "") == "notis_peft" and n.lineno <= late.lineno <= n.end_lineno @@ -262,7 +264,8 @@ def test_the_adapter_load_keeps_the_callers_revision(): for class_name in ("FastLanguageModel", "FastModel"): function = _function(tree, "from_pretrained", class_name) peft_loads = [ - c for c in _calls(function, "from_pretrained") + c + for c in _calls(function, "from_pretrained") if ast.unparse(c.func).startswith("PeftModel") ] assert peft_loads, f"{class_name} has no PeftModel load" @@ -280,15 +283,18 @@ def test_a_pinned_load_does_not_mix_refs_with_vllm(path, flag): name = "FastLlamaModel" if path is LLAMA else "FastBaseModel" function = _function(tree, "from_pretrained", name) clears = [ - n for n in ast.walk(function) + n + for n in ast.walk(function) if isinstance(n, ast.Assign) and any(getattr(t, "id", None) == flag for t in n.targets) - and isinstance(n.value, ast.Constant) and n.value.value is None + and isinstance(n.value, ast.Constant) + and n.value.value is None ] assert clears, f"{path.name} never drops the revision on the vLLM path" # It must happen before the config load, or the config is pinned and the weights are not. configs = [ - c for c in _calls(function, "from_pretrained") + c + for c in _calls(function, "from_pretrained") if ast.unparse(c.func).startswith("AutoConfig") ] assert configs From d2726a9ef081dc6154d958935766f97d1b9481c3 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 2 Aug 2026 15:31:23 +0000 Subject: [PATCH 09/20] Keep the revision when vLLM was requested but is unavailable The vLLM guard sat at the end of the same block that turns fast_inference off when vLLM is missing or the GPU is older than sm70. In that case the load falls through in-process and can honour the revision, but the guard dropped it anyway. Re-check fast_inference in the condition. --- tests/python/test_revision_forwarding.py | 25 ++++++++++++++++++++++++ unsloth/models/llama.py | 4 +++- 2 files changed, 28 insertions(+), 1 deletion(-) diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py index 7e0ec45a7e..5ece03d0b7 100644 --- a/tests/python/test_revision_forwarding.py +++ b/tests/python/test_revision_forwarding.py @@ -316,3 +316,28 @@ def test_local_snapshot_resolution_takes_the_revision(): wrapper = _function(tree, "_hub_repo_or_local_path") inner = _calls(wrapper, "_resolve_hub_repo_local_dir") assert inner and all(_revision_kwarg(c) is not None for c in inner) + + +def test_the_vllm_drop_re_checks_fast_inference(): + """`fast_inference` is turned off in that same block when vLLM is missing or the GPU + is too old, and the in-process load that then runs can honour the revision.""" + function = _function(_tree(LLAMA), "from_pretrained", "FastLlamaModel") + clears = [ + n + for n in ast.walk(function) + if isinstance(n, ast.Assign) + and any(getattr(t, "id", None) == "revision" for t in n.targets) + and isinstance(n.value, ast.Constant) + and n.value.value is None + ] + assert clears, "the vLLM revision drop is gone" + for clear in clears: + guards = [ + n + for n in ast.walk(function) + if isinstance(n, ast.If) + and n.lineno <= clear.lineno <= n.end_lineno + and "fast_inference" in ast.unparse(n.test) + and "revision" in ast.unparse(n.test) + ] + assert guards, "the drop must re-check fast_inference, not just the enclosing block" diff --git a/unsloth/models/llama.py b/unsloth/models/llama.py index 3de5b5f8c5..cb8b96a5aa 100644 --- a/unsloth/models/llama.py +++ b/unsloth/models/llama.py @@ -2338,7 +2338,9 @@ def from_pretrained( raise RuntimeError( "Unsloth: `unsloth_vllm_standby` is True, but environment variable `UNSLOTH_VLLM_STANDBY` is not set to 1!" ) - if revision is not None: + # `fast_inference` may have just been turned off above, and the in-process + # load that then runs can honour the revision, so re-check it here. + if fast_inference and revision is not None: # load_vllm takes no revision, so vLLM fetches the default branch. Pinning # only the config and tokenizer would mix two refs in one model. logger.warning_once( From d6a354add8e09811571212fa64630d3127843264 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 2 Aug 2026 15:50:13 +0000 Subject: [PATCH 10/20] Keep the pin where the load can honour it, and tailor the warning Three more from review: - A num_labels load goes through AutoModelForSequenceClassification in-process no matter what fast_inference says, so the vLLM guard was discarding a revision the load could have used. Condition it on the same `fast_inference and num_labels is None` predicate the prefetch warm already uses. - use_exact_model_name only gates the mapper substitution. The ModelScope download, the ALLOW_PREQUANTIZED_MODELS strip and fast_inference_setup ignore it, so the warning was sending callers round the same loop. Record whether the mapper is what moved the name and only offer the remedy then. - The tokenizer does not always come from the base model's repo. Loading a PEFT repo with an explicit tokenizer_name pointing at the adapter dropped the pin for the tokenizer while PeftModel loaded the adapter from the requested ref, mixing two refs. _revision_for_tokenizer_repo now resolves it where the repos are known and both dispatches carry it, replacing the tokenizer_name == model_name guess in llama.py and vision.py. vision.py pops it from kwargs, since the weight load forwards **kwargs and transformers has no such argument. Seven new tests, all failing before this. --- tests/python/test_revision_forwarding.py | 97 ++++++++++++++++++++++-- unsloth/models/llama.py | 13 ++-- unsloth/models/loader.py | 55 ++++++++++++-- unsloth/models/vision.py | 9 ++- 4 files changed, 149 insertions(+), 25 deletions(-) diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py index 5ece03d0b7..017fec3746 100644 --- a/tests/python/test_revision_forwarding.py +++ b/tests/python/test_revision_forwarding.py @@ -192,7 +192,7 @@ def test_no_revision_stays_none_even_when_remapped(): assert gate(None, "unsloth/llama-3-8b-bnb-4bit", "meta-llama/Meta-Llama-3-8B") is None -def test_the_gate_warns_exactly_once_when_it_drops_a_revision(): +def _gate_with_warnings(): source = LOADER.read_text(encoding = "utf-8") function = _function(ast.parse(source), "_revision_for_resolved_repo") warnings = [] @@ -200,13 +200,43 @@ def test_the_gate_warns_exactly_once_when_it_drops_a_revision(): module = ast.Module(body = [function], type_ignores = []) ast.fix_missing_locations(module) exec(compile(module, str(LOADER), "exec"), namespace) + return namespace["_revision_for_resolved_repo"], warnings + - namespace["_revision_for_resolved_repo"]("abc123", "unsloth/x-bnb-4bit", "org/x") +def test_the_gate_warns_exactly_once_when_it_drops_a_revision(): + gate, warnings = _gate_with_warnings() + gate("abc123", "unsloth/x-bnb-4bit", "org/x", True) assert len(warnings) == 1 message = warnings[0] # Both repos have to be named or the user cannot tell which load was silently redirected. assert "abc123" in message and "org/x" in message and "unsloth/x-bnb-4bit" in message - assert "use_exact_model_name" in message + + +def test_exact_name_mode_is_only_offered_when_it_would_help(): + """It gates the mapper substitution alone. The ModelScope download, the + ALLOW_PREQUANTIZED_MODELS strip and fast_inference_setup all ignore it, so + recommending it there sends the caller round the same loop.""" + gate, warnings = _gate_with_warnings() + gate("abc123", "unsloth/x-bnb-4bit", "org/x", True) + assert "use_exact_model_name" in warnings[0] + + gate, warnings = _gate_with_warnings() + gate("abc123", "/tmp/modelscope/x", "org/x", False) + assert "use_exact_model_name" not in warnings[0] + + +def test_both_loader_paths_pass_the_mapper_flag(): + tree = _tree(LOADER) + for class_name in ("FastLanguageModel", "FastModel"): + function = _function(tree, "from_pretrained", class_name) + assert [ + n for n in ast.walk(function) + if isinstance(n, ast.Assign) + and any(getattr(t, "id", None) == "mapper_moved_name" for t in n.targets) + ], f"{class_name} must record whether the mapper moved the name" + for call in _calls(function, "_revision_for_resolved_repo"): + names = [getattr(a, "id", None) for a in call.args] + assert "mapper_moved_name" in names, "the gate needs the flag to tailor its remedy" def test_both_loader_paths_gate_before_and_after_resolution(): @@ -318,9 +348,10 @@ def test_local_snapshot_resolution_takes_the_revision(): assert inner and all(_revision_kwarg(c) is not None for c in inner) -def test_the_vllm_drop_re_checks_fast_inference(): - """`fast_inference` is turned off in that same block when vLLM is missing or the GPU - is too old, and the in-process load that then runs can honour the revision.""" +def test_the_vllm_drop_only_fires_when_vllm_owns_the_weights(): + """fast_inference is turned off in that same block when vLLM is missing or the GPU + is too old, and a num_labels load goes through transformers regardless. Both of + those can honour the pin, so the drop must not be unconditional.""" function = _function(_tree(LLAMA), "from_pretrained", "FastLlamaModel") clears = [ n @@ -337,7 +368,57 @@ def test_the_vllm_drop_re_checks_fast_inference(): for n in ast.walk(function) if isinstance(n, ast.If) and n.lineno <= clear.lineno <= n.end_lineno - and "fast_inference" in ast.unparse(n.test) and "revision" in ast.unparse(n.test) ] - assert guards, "the drop must re-check fast_inference, not just the enclosing block" + assert guards, "the drop needs its own condition" + test = ast.unparse(guards[0].test) + assert "fast_inference" in test, "must re-check fast_inference" + assert "num_labels" in test, "a num_labels load runs in-process and can be pinned" + + +def test_the_tokenizer_revision_is_resolved_by_the_loader(): + """The tokenizer repo is not always the base model's: a PEFT load whose + tokenizer_name is the adapter keeps the caller's ref, which the base model cannot.""" + tree = _tree(LOADER) + helper = _function(tree, "_revision_for_tokenizer_repo") + assert helper, "the loader must resolve the tokenizer repo's revision" + for class_name in ("FastLanguageModel", "FastModel"): + function = _function(tree, "from_pretrained", class_name) + dispatches = [ + c for c in _calls(function, "from_pretrained") + if any(k.arg == "tokenizer_revision" for k in c.keywords) + ] + assert dispatches, f"{class_name} must dispatch a tokenizer_revision" + for call in dispatches: + keyword = next(k for k in call.keywords if k.arg == "tokenizer_revision") + assert isinstance(keyword.value, ast.Call), "it has to be the resolved value" + + +def test_llama_uses_the_dispatched_tokenizer_revision(): + function = _function(_tree(LLAMA), "from_pretrained", "FastLlamaModel") + assert "tokenizer_revision" in _params(function) + loads = _calls(function, "load_correct_tokenizer") + assert loads + for call in loads: + keyword = _revision_kwarg(call) + assert keyword is not None + assert getattr(keyword.value, "id", None) == "tokenizer_revision" + + +def test_vision_pops_the_tokenizer_revision_before_the_weight_load(): + """FastBaseModel forwards **kwargs to the weight load, and transformers has no + tokenizer_revision argument, so it must be popped rather than read.""" + function = _function(_tree(VISION), "from_pretrained", "FastBaseModel") + pops = [ + c for c in ast.walk(function) + if isinstance(c, ast.Call) + and ast.unparse(c.func).endswith("kwargs.pop") + and c.args and getattr(c.args[0], "value", None) == "tokenizer_revision" + ] + assert pops, "tokenizer_revision must be popped from kwargs" + weight_loads = [ + c for c in _calls(function, "from_pretrained") + if ast.unparse(c.func).startswith("auto_model") + ] + assert weight_loads + assert pops[0].lineno < min(c.lineno for c in weight_loads) diff --git a/unsloth/models/llama.py b/unsloth/models/llama.py index cb8b96a5aa..01585a5a52 100644 --- a/unsloth/models/llama.py +++ b/unsloth/models/llama.py @@ -2299,6 +2299,7 @@ def from_pretrained( tokenizer_name = None, trust_remote_code = False, revision = None, + tokenizer_revision = None, fast_inference = False, # uses vLLM gpu_memory_utilization = 0.5, float8_kv_cache = False, @@ -2338,9 +2339,10 @@ def from_pretrained( raise RuntimeError( "Unsloth: `unsloth_vllm_standby` is True, but environment variable `UNSLOTH_VLLM_STANDBY` is not set to 1!" ) - # `fast_inference` may have just been turned off above, and the in-process - # load that then runs can honour the revision, so re-check it here. - if fast_inference and revision is not None: + # Only vLLM cannot take a revision. fast_inference may have just been turned + # off above, and a num_labels load goes in-process regardless; both of those + # can honour the pin, so use the same predicate as the prefetch warm below. + if fast_inference and num_labels is None and revision is not None: # load_vllm takes no revision, so vLLM fetches the default branch. Pinning # only the config and tokenizer would mix two refs in one model. logger.warning_once( @@ -2506,8 +2508,7 @@ def from_pretrained( cache_dir = _tokenizer_cache_dir, local_files_only = kwargs.get("local_files_only", False), tokenizer_only = True, - # Matches the tokenizer load below, which only pins its own repo. - revision = revision if _tokenizer_repo == model_name else None, + revision = tokenizer_revision, ) has_rope_scaling = False @@ -2799,7 +2800,7 @@ def from_pretrained( token = token, trust_remote_code = trust_remote_code, fix_tokenizer = fix_tokenizer, - revision = revision if tokenizer_name == model_name else None, + revision = tokenizer_revision, **_tokenizer_cache_kwargs, ) diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index ce2c335ec7..3a553c8fc8 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -149,23 +149,44 @@ def _strip_unsloth_bnb_4bit_suffix(model_name: str) -> str: return s -def _revision_for_resolved_repo(revision, model_name, old_model_name): +def _revision_for_resolved_repo(revision, model_name, old_model_name, mapper_moved_name = False): """Drop `revision` once the requested repo has been remapped to another one. A revision names a branch/tag/SHA on the repo the caller asked for, but from_pretrained may resolve model_name to a different repo (a pre-quantized mirror, an fp8 temp dir, a - ModelScope snapshot, a -bnb-4bit strip), where that ref does not exist. + ModelScope snapshot, a -bnb-4bit strip), where that ref does not exist. Only the mapper + substitution answers to use_exact_model_name, so only suggest it when it would help. """ if revision is None or model_name == old_model_name: return revision + remedy = ( + " Pass `use_exact_model_name = True` to load your repo as-is." + if mapper_moved_name else "" + ) logger.warning_once( f"Unsloth: Ignoring revision = `{revision}` since `{old_model_name}` resolved to " - f"`{model_name}`, which does not have that revision. " - "Pass `use_exact_model_name = True` to load your repo as-is." + f"`{model_name}`, which does not have that revision.{remedy}" ) return None +def _revision_for_tokenizer_repo( + tokenizer_name, model_name, old_model_name, revision, base_revision +): + """Pick the revision for whichever repo the tokenizer is actually read from. + + It is not always the base model's: a PEFT load keeps the caller's repo when + tokenizer_name points at the adapter, while an unset tokenizer_name follows the + resolved model_name and so follows the gated revision. + """ + repo = tokenizer_name if tokenizer_name else model_name + if repo == old_model_name: + return revision + if repo == model_name: + return base_revision + return None + + def _config_get( config, field_name, @@ -554,6 +575,8 @@ def from_pretrained( # on-the-fly quantization to avoid double quantization if load_in_fp8 != False and new_model_name != old_model_name: load_in_fp8 = False + # Only this block honours use_exact_model_name; the transforms below do not. + mapper_moved_name = model_name != old_model_name # Check if pre-quantized models are allowed # AMD Instinct GPUs need blocksize = 128 on bitsandbytes < 0.49.2 (our pre-quants use blocksize = 64) @@ -576,7 +599,9 @@ def from_pretrained( # Gate before the probe below, or a pinned 4bit load fails against the mirror # instead of warning. Kept separate: `revision` still names the adapter repo. - base_revision = _revision_for_resolved_repo(revision, model_name, old_model_name) + base_revision = _revision_for_resolved_repo( + revision, model_name, old_model_name, mapper_moved_name + ) # First check if it's a normal model via AutoConfig from huggingface_hub.utils import ( @@ -874,7 +899,9 @@ def from_pretrained( # model_name can move once more here. Skip for PEFT: model_name is then the base # model, and `revision` names the adapter that PeftModel.from_pretrained loads below. if not is_peft: - base_revision = _revision_for_resolved_repo(base_revision, model_name, old_model_name) + base_revision = _revision_for_resolved_repo( + base_revision, model_name, old_model_name, mapper_moved_name + ) load_in_4bit_kwargs = load_in_4bit load_in_8bit_kwargs = load_in_8bit @@ -903,6 +930,9 @@ def from_pretrained( tokenizer_name = tokenizer_name, trust_remote_code = trust_remote_code, revision = base_revision if not is_peft else None, + tokenizer_revision = _revision_for_tokenizer_repo( + tokenizer_name, model_name, old_model_name, revision, base_revision + ), fast_inference = fast_inference, gpu_memory_utilization = gpu_memory_utilization, float8_kv_cache = float8_kv_cache, @@ -1281,6 +1311,8 @@ def from_pretrained( # on-the-fly quantization to avoid double quantization if load_in_fp8 != False and new_model_name != old_model_name: load_in_fp8 = False + # Only this block honours use_exact_model_name; the transforms below do not. + mapper_moved_name = model_name != old_model_name # Check if pre-quantized models are allowed # AMD Instinct GPUs need blocksize = 128 on bitsandbytes < 0.49.2 (our pre-quants use blocksize = 64) @@ -1304,7 +1336,9 @@ def from_pretrained( # Gate before the probe below, or a pinned 4bit load fails against the mirror # instead of warning. Kept separate: `revision` still names the adapter repo. - base_revision = _revision_for_resolved_repo(revision, model_name, old_model_name) + base_revision = _revision_for_resolved_repo( + revision, model_name, old_model_name, mapper_moved_name + ) # First check if it's a normal model via AutoConfig from huggingface_hub.utils import ( @@ -1831,7 +1865,9 @@ def _dispatch_diffusion(): # FastBaseModel remaps again via fast_inference_setup. Skip for PEFT: model_name is # then the base model, and `revision` names the adapter PeftModel loads below. if not is_peft: - base_revision = _revision_for_resolved_repo(base_revision, model_name, old_model_name) + base_revision = _revision_for_resolved_repo( + base_revision, model_name, old_model_name, mapper_moved_name + ) model, tokenizer = FastBaseModel.from_pretrained( model_name = model_name, @@ -1845,6 +1881,9 @@ def _dispatch_diffusion(): device_map = device_map, trust_remote_code = trust_remote_code, revision = base_revision if not is_peft else None, + tokenizer_revision = _revision_for_tokenizer_repo( + tokenizer_name, model_name, old_model_name, revision, base_revision + ), model_types = model_types, tokenizer_name = tokenizer_name, auto_model = auto_model, diff --git a/unsloth/models/vision.py b/unsloth/models/vision.py index cab28d9794..e5c63a0187 100644 --- a/unsloth/models/vision.py +++ b/unsloth/models/vision.py @@ -871,6 +871,7 @@ def from_pretrained( # Read revision from kwargs, not the signature: the weight load below forwards # **kwargs, so binding it would drop it there. Pin its repo before any remap. _revision = kwargs.get("revision") + _tokenizer_revision_arg = kwargs.pop("tokenizer_revision", None) if _revision is not None and fast_inference and is_vLLM_available(): # load_vllm takes no revision, so vLLM fetches the default branch. Pinning only # the config and tokenizer would mix two refs in one model. @@ -1458,9 +1459,11 @@ def from_pretrained( # Counteract saved tokenizers tokenizer_name = model_name if tokenizer_name is None else tokenizer_name - # The tokenizer repo can differ from the revision's repo (a caller override, or - # model_name remapped by fast_inference_setup above). - _tokenizer_revision = _revision if tokenizer_name == _revision_repo else None + # Resolved by the loader, which knows whether the tokenizer repo is the caller's + # (a PEFT adapter) or the resolved base model. Falls back for a direct call. + _tokenizer_revision = _tokenizer_revision_arg + if _tokenizer_revision is None and tokenizer_name == _revision_repo: + _tokenizer_revision = _revision # On the vLLM path the tokenizer warm was deferred (fast_inference_setup may remap model_name). # Warm the now-final tokenizer repo so the load below hits the cache (a cached/local repo is a no-op). From 1607b3d46b124591f9d5bf5d52ba2ea3b345daec Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Sun, 2 Aug 2026 15:50:57 +0000 Subject: [PATCH 11/20] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- tests/python/test_revision_forwarding.py | 15 ++++++++++----- unsloth/models/loader.py | 10 +++++++--- 2 files changed, 17 insertions(+), 8 deletions(-) diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py index 017fec3746..76683fc501 100644 --- a/tests/python/test_revision_forwarding.py +++ b/tests/python/test_revision_forwarding.py @@ -230,7 +230,8 @@ def test_both_loader_paths_pass_the_mapper_flag(): for class_name in ("FastLanguageModel", "FastModel"): function = _function(tree, "from_pretrained", class_name) assert [ - n for n in ast.walk(function) + n + for n in ast.walk(function) if isinstance(n, ast.Assign) and any(getattr(t, "id", None) == "mapper_moved_name" for t in n.targets) ], f"{class_name} must record whether the mapper moved the name" @@ -385,7 +386,8 @@ def test_the_tokenizer_revision_is_resolved_by_the_loader(): for class_name in ("FastLanguageModel", "FastModel"): function = _function(tree, "from_pretrained", class_name) dispatches = [ - c for c in _calls(function, "from_pretrained") + c + for c in _calls(function, "from_pretrained") if any(k.arg == "tokenizer_revision" for k in c.keywords) ] assert dispatches, f"{class_name} must dispatch a tokenizer_revision" @@ -410,14 +412,17 @@ def test_vision_pops_the_tokenizer_revision_before_the_weight_load(): tokenizer_revision argument, so it must be popped rather than read.""" function = _function(_tree(VISION), "from_pretrained", "FastBaseModel") pops = [ - c for c in ast.walk(function) + c + for c in ast.walk(function) if isinstance(c, ast.Call) and ast.unparse(c.func).endswith("kwargs.pop") - and c.args and getattr(c.args[0], "value", None) == "tokenizer_revision" + and c.args + and getattr(c.args[0], "value", None) == "tokenizer_revision" ] assert pops, "tokenizer_revision must be popped from kwargs" weight_loads = [ - c for c in _calls(function, "from_pretrained") + c + for c in _calls(function, "from_pretrained") if ast.unparse(c.func).startswith("auto_model") ] assert weight_loads diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index 3a553c8fc8..a25f2273f4 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -149,7 +149,12 @@ def _strip_unsloth_bnb_4bit_suffix(model_name: str) -> str: return s -def _revision_for_resolved_repo(revision, model_name, old_model_name, mapper_moved_name = False): +def _revision_for_resolved_repo( + revision, + model_name, + old_model_name, + mapper_moved_name = False, +): """Drop `revision` once the requested repo has been remapped to another one. A revision names a branch/tag/SHA on the repo the caller asked for, but from_pretrained @@ -160,8 +165,7 @@ def _revision_for_resolved_repo(revision, model_name, old_model_name, mapper_mov if revision is None or model_name == old_model_name: return revision remedy = ( - " Pass `use_exact_model_name = True` to load your repo as-is." - if mapper_moved_name else "" + " Pass `use_exact_model_name = True` to load your repo as-is." if mapper_moved_name else "" ) logger.warning_once( f"Unsloth: Ignoring revision = `{revision}` since `{old_model_name}` resolved to " From 61baa18011767e6c0344a3a5816140d1f235a910 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 2 Aug 2026 16:09:33 +0000 Subject: [PATCH 12/20] Keep the adapter ref off the base tokenizer, and pin both or neither Three more from review, all fallout from splitting tokenizer_revision out: - Skipping the late gate for PEFT leaves base_revision naming the adapter, and a remote PEFT load without an explicit tokenizer_name reads its tokenizer from the base repo, so that ref was handed to the wrong repository. Both dispatches now derive one model_revision and pass it to the base load and to the tokenizer resolution alike, so the base tokenizer can only ever get the base model's ref. - FastLlamaModel is exported, and the architecture wrappers forward `revision` through **kwargs without the new internal tokenizer_revision, so a direct call pinned the config and weights while the tokenizer read the default branch. Fall back to `revision` when the tokenizer repo is the model repo, before the warm so it does not fetch the wrong ref either. - The vLLM guard cleared only the model pin, leaving vLLM on the default branch with the tokenizer still on the requested ref. Clear both, in llama.py and in the parallel FastBaseModel block. Seven new tests. --- tests/python/test_revision_forwarding.py | 92 ++++++++++++++++++++++++ unsloth/models/llama.py | 6 ++ unsloth/models/loader.py | 22 +++--- unsloth/models/vision.py | 1 + 4 files changed, 112 insertions(+), 9 deletions(-) diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py index 76683fc501..66af642615 100644 --- a/tests/python/test_revision_forwarding.py +++ b/tests/python/test_revision_forwarding.py @@ -427,3 +427,95 @@ def test_vision_pops_the_tokenizer_revision_before_the_weight_load(): ] assert weight_loads assert pops[0].lineno < min(c.lineno for c in weight_loads) + + +def _load_tokenizer_gate(): + source = LOADER.read_text(encoding = "utf-8") + function = _function(ast.parse(source), "_revision_for_tokenizer_repo") + namespace = {} + module = ast.Module(body = [function], type_ignores = []) + ast.fix_missing_locations(module) + exec(compile(module, str(LOADER), "exec"), namespace) + return namespace["_revision_for_tokenizer_repo"] + + +def test_an_adapter_ref_never_reaches_the_base_tokenizer(): + """On a PEFT load the late gate is skipped, so the gated value still names the + adapter. The base repo's tokenizer must take the model load's ref, which is None.""" + gate = _load_tokenizer_gate() + # Remote adapter, no explicit tokenizer_name: the tokenizer follows the base model. + assert gate(None, "org/base", "org/adapter", "v2", None) is None + + +def test_an_adapter_hosted_tokenizer_keeps_the_callers_ref(): + gate = _load_tokenizer_gate() + assert gate("org/adapter", "org/base", "org/adapter", "v2", None) == "v2" + + +def test_a_plain_load_gives_the_tokenizer_the_model_ref(): + gate = _load_tokenizer_gate() + assert gate(None, "org/model", "org/model", "v2", "v2") == "v2" + # A third-party tokenizer repo is pinned by neither. + assert gate("other/tok", "org/model", "org/model", "v2", "v2") is None + + +def test_both_dispatches_share_one_model_revision(): + """The value handed to the base load and the one the tokenizer resolution sees have + to be the same, or the PEFT case leaks the adapter ref into the base repo.""" + tree = _tree(LOADER) + for class_name in ("FastLanguageModel", "FastModel"): + function = _function(tree, "from_pretrained", class_name) + assert [ + n for n in ast.walk(function) + if isinstance(n, ast.Assign) + and any(getattr(t, "id", None) == "model_revision" for t in n.targets) + ], f"{class_name} must derive one model_revision" + dispatch = next( + c for c in _calls(function, "from_pretrained") + if any(k.arg == "tokenizer_revision" for k in c.keywords) + ) + model_kw = _revision_kwarg(dispatch) + assert getattr(model_kw.value, "id", None) == "model_revision" + tok_kw = next(k for k in dispatch.keywords if k.arg == "tokenizer_revision") + assert "model_revision" in ast.unparse(tok_kw.value) + + +def test_a_direct_llama_call_still_pins_its_tokenizer(): + """FastLlamaModel is exported and the architecture wrappers forward only `revision`, + so tokenizer_revision has to fall back to it when the repos are the same.""" + function = _function(_tree(LLAMA), "from_pretrained", "FastLlamaModel") + fallbacks = [ + n for n in ast.walk(function) + if isinstance(n, ast.Assign) + and any(getattr(t, "id", None) == "tokenizer_revision" for t in n.targets) + and getattr(n.value, "id", None) == "revision" + ] + assert fallbacks, "no fallback from revision to tokenizer_revision" + warms = _calls(function, "maybe_prefetch_hf_snapshot") + tokenizer_warms = [ + c for c in warms + if any(k.arg == "revision" and getattr(k.value, "id", None) == "tokenizer_revision" + for k in c.keywords) + ] + assert tokenizer_warms, "the tokenizer warm should use the same pin" + # The fallback must precede the warm, or the warm fetches the wrong ref. + assert fallbacks[0].lineno < min(c.lineno for c in tokenizer_warms) + + +@pytest.mark.parametrize( + "path, cls, name", + [(LLAMA, "FastLlamaModel", "tokenizer_revision"), + (VISION, "FastBaseModel", "_tokenizer_revision_arg")], +) +def test_the_vllm_drop_clears_the_tokenizer_pin_too(path, cls, name): + """Clearing only the model pin left vLLM on the default branch while the tokenizer + stayed on the requested ref.""" + function = _function(ast.parse(path.read_text(encoding = "utf-8")), "from_pretrained", cls) + clears = [ + n for n in ast.walk(function) + if isinstance(n, ast.Assign) + and any(getattr(t, "id", None) == name for t in n.targets) + and isinstance(n.value, ast.Constant) + and n.value.value is None + ] + assert clears, f"{path.name} never clears {name} on the vLLM path" diff --git a/unsloth/models/llama.py b/unsloth/models/llama.py index 01585a5a52..cc0965d54a 100644 --- a/unsloth/models/llama.py +++ b/unsloth/models/llama.py @@ -2350,6 +2350,12 @@ def from_pretrained( "the default branch. Use `fast_inference = False` to load a pinned revision." ) revision = None + tokenizer_revision = None + + if tokenizer_revision is None and tokenizer_name in (None, model_name): + # A direct FastLlamaModel call, or an architecture wrapper forwarding only + # `revision`, leaves this unset while the config and weights are pinned. + tokenizer_revision = revision token = hf_login(token) if model_patcher is None: diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index a25f2273f4..0f898cf99f 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -175,19 +175,19 @@ def _revision_for_resolved_repo( def _revision_for_tokenizer_repo( - tokenizer_name, model_name, old_model_name, revision, base_revision + tokenizer_name, model_name, old_model_name, revision, model_revision ): """Pick the revision for whichever repo the tokenizer is actually read from. - It is not always the base model's: a PEFT load keeps the caller's repo when - tokenizer_name points at the adapter, while an unset tokenizer_name follows the - resolved model_name and so follows the gated revision. + It is not always the base model's: an adapter-hosted tokenizer keeps the caller's ref, + while an unset tokenizer_name follows the resolved model_name and so takes whatever the + model load itself uses (None on a PEFT load, whose ref belongs to the adapter). """ repo = tokenizer_name if tokenizer_name else model_name if repo == old_model_name: return revision if repo == model_name: - return base_revision + return model_revision return None @@ -906,6 +906,8 @@ def from_pretrained( base_revision = _revision_for_resolved_repo( base_revision, model_name, old_model_name, mapper_moved_name ) + # On a PEFT load model_name is the base model, which the caller's ref is not for. + model_revision = base_revision if not is_peft else None load_in_4bit_kwargs = load_in_4bit load_in_8bit_kwargs = load_in_8bit @@ -933,9 +935,9 @@ def from_pretrained( model_patcher = dispatch_model, tokenizer_name = tokenizer_name, trust_remote_code = trust_remote_code, - revision = base_revision if not is_peft else None, + revision = model_revision, tokenizer_revision = _revision_for_tokenizer_repo( - tokenizer_name, model_name, old_model_name, revision, base_revision + tokenizer_name, model_name, old_model_name, revision, model_revision ), fast_inference = fast_inference, gpu_memory_utilization = gpu_memory_utilization, @@ -1872,6 +1874,8 @@ def _dispatch_diffusion(): base_revision = _revision_for_resolved_repo( base_revision, model_name, old_model_name, mapper_moved_name ) + # On a PEFT load model_name is the base model, which the caller's ref is not for. + model_revision = base_revision if not is_peft else None model, tokenizer = FastBaseModel.from_pretrained( model_name = model_name, @@ -1884,9 +1888,9 @@ def _dispatch_diffusion(): token = token, device_map = device_map, trust_remote_code = trust_remote_code, - revision = base_revision if not is_peft else None, + revision = model_revision, tokenizer_revision = _revision_for_tokenizer_repo( - tokenizer_name, model_name, old_model_name, revision, base_revision + tokenizer_name, model_name, old_model_name, revision, model_revision ), model_types = model_types, tokenizer_name = tokenizer_name, diff --git a/unsloth/models/vision.py b/unsloth/models/vision.py index e5c63a0187..ca67259c90 100644 --- a/unsloth/models/vision.py +++ b/unsloth/models/vision.py @@ -880,6 +880,7 @@ def from_pretrained( "the default branch. Use `fast_inference = False` to load a pinned revision." ) _revision = None + _tokenizer_revision_arg = None kwargs.pop("revision", None) _revision_repo = model_name From 0f0517eaba713913d7b3e6051ce52756b9e2a4f7 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Sun, 2 Aug 2026 16:11:00 +0000 Subject: [PATCH 13/20] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- tests/python/test_revision_forwarding.py | 27 ++++++++++++++++-------- 1 file changed, 18 insertions(+), 9 deletions(-) diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py index 66af642615..3cebdc79e9 100644 --- a/tests/python/test_revision_forwarding.py +++ b/tests/python/test_revision_forwarding.py @@ -466,12 +466,14 @@ def test_both_dispatches_share_one_model_revision(): for class_name in ("FastLanguageModel", "FastModel"): function = _function(tree, "from_pretrained", class_name) assert [ - n for n in ast.walk(function) + n + for n in ast.walk(function) if isinstance(n, ast.Assign) and any(getattr(t, "id", None) == "model_revision" for t in n.targets) ], f"{class_name} must derive one model_revision" dispatch = next( - c for c in _calls(function, "from_pretrained") + c + for c in _calls(function, "from_pretrained") if any(k.arg == "tokenizer_revision" for k in c.keywords) ) model_kw = _revision_kwarg(dispatch) @@ -485,7 +487,8 @@ def test_a_direct_llama_call_still_pins_its_tokenizer(): so tokenizer_revision has to fall back to it when the repos are the same.""" function = _function(_tree(LLAMA), "from_pretrained", "FastLlamaModel") fallbacks = [ - n for n in ast.walk(function) + n + for n in ast.walk(function) if isinstance(n, ast.Assign) and any(getattr(t, "id", None) == "tokenizer_revision" for t in n.targets) and getattr(n.value, "id", None) == "revision" @@ -493,9 +496,12 @@ def test_a_direct_llama_call_still_pins_its_tokenizer(): assert fallbacks, "no fallback from revision to tokenizer_revision" warms = _calls(function, "maybe_prefetch_hf_snapshot") tokenizer_warms = [ - c for c in warms - if any(k.arg == "revision" and getattr(k.value, "id", None) == "tokenizer_revision" - for k in c.keywords) + c + for c in warms + if any( + k.arg == "revision" and getattr(k.value, "id", None) == "tokenizer_revision" + for k in c.keywords + ) ] assert tokenizer_warms, "the tokenizer warm should use the same pin" # The fallback must precede the warm, or the warm fetches the wrong ref. @@ -504,15 +510,18 @@ def test_a_direct_llama_call_still_pins_its_tokenizer(): @pytest.mark.parametrize( "path, cls, name", - [(LLAMA, "FastLlamaModel", "tokenizer_revision"), - (VISION, "FastBaseModel", "_tokenizer_revision_arg")], + [ + (LLAMA, "FastLlamaModel", "tokenizer_revision"), + (VISION, "FastBaseModel", "_tokenizer_revision_arg"), + ], ) def test_the_vllm_drop_clears_the_tokenizer_pin_too(path, cls, name): """Clearing only the model pin left vLLM on the default branch while the tokenizer stayed on the requested ref.""" function = _function(ast.parse(path.read_text(encoding = "utf-8")), "from_pretrained", cls) clears = [ - n for n in ast.walk(function) + n + for n in ast.walk(function) if isinstance(n, ast.Assign) and any(getattr(t, "id", None) == name for t in n.targets) and isinstance(n.value, ast.Constant) From cea926e0587c1f3874f0d095e66c317c0cec70c0 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 2 Aug 2026 16:41:56 +0000 Subject: [PATCH 14/20] Keep one ref per repo on the fp8, vLLM config and tokenizer paths Four ways a pin could still land on the wrong ref: - A plain load that names its own repo as tokenizer_name kept the caller's revision even after a remap had already dropped it off the config and weights, so mirror weights paired with a pinned tokenizer. Only a PEFT adapter is a genuinely separate repo, so only it keeps that ref now. - FastModel probes the config before dispatching and FastBaseModel skips its own load while that config is set, so the vLLM path received a config read at the pinned ref alongside the default-branch weights vLLM fetches. The probed config is now withheld there; a caller's own config still goes down. - The get_auto_processor fallback under AutoProcessor ran unpinned. - _offline_quantize_to_fp8 read the default branch and cached under a name that ignored the revision, so load_in_fp8 with a revision quantized the wrong ref and could reuse another ref's artifact. --- tests/python/test_revision_forwarding.py | 161 ++++++++++++++++++++++- unsloth/models/loader.py | 46 +++++-- unsloth/models/loader_utils.py | 13 +- unsloth/models/vision.py | 1 + 4 files changed, 208 insertions(+), 13 deletions(-) diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py index 3cebdc79e9..0ae9cb3dfb 100644 --- a/tests/python/test_revision_forwarding.py +++ b/tests/python/test_revision_forwarding.py @@ -448,8 +448,18 @@ def test_an_adapter_ref_never_reaches_the_base_tokenizer(): def test_an_adapter_hosted_tokenizer_keeps_the_callers_ref(): + """An adapter is a separate repo with its own history, so the caller's ref still + names it even though the base model it sits on cannot answer to it.""" gate = _load_tokenizer_gate() - assert gate("org/adapter", "org/base", "org/adapter", "v2", None) == "v2" + assert gate("org/adapter", "org/base", "org/adapter", "v2", None, True) == "v2" + + +def test_a_remapped_plain_load_drops_the_tokenizer_pin_too(): + """Naming the requested repo as tokenizer_name must not smuggle the ref back in: the + weights now come off a mirror's default branch, and a pinned tokenizer beside them is + the ref mismatch the gate exists to prevent. Only a PEFT adapter is a separate repo.""" + gate = _load_tokenizer_gate() + assert gate("org/model", "unsloth/model-bnb-4bit", "org/model", "v2", None) is None def test_a_plain_load_gives_the_tokenizer_the_model_ref(): @@ -528,3 +538,152 @@ def test_the_vllm_drop_clears_the_tokenizer_pin_too(path, cls, name): and n.value.value is None ] assert clears, f"{path.name} never clears {name} on the vLLM path" + + +def _simulate_loader(): + """Run the loader's two revision decisions the way from_pretrained sequences them.""" + tree = ast.parse(LOADER.read_text(encoding = "utf-8")) + namespace = {"logger": types.SimpleNamespace(warning_once = lambda *a, **k: None)} + functions = [ + n for n in tree.body + if isinstance(n, ast.FunctionDef) + and n.name in ("_revision_for_resolved_repo", "_revision_for_tokenizer_repo") + ] + module = ast.Module(body = functions, type_ignores = []) + ast.fix_missing_locations(module) + exec(compile(module, str(LOADER), "exec"), namespace) + gate = namespace["_revision_for_resolved_repo"] + tokenizer_gate = namespace["_revision_for_tokenizer_repo"] + + def run(old_model_name, model_name, is_peft, tokenizer_name, revision, mapper_moved_name): + base_revision = gate(revision, model_name, old_model_name, mapper_moved_name) + if not is_peft: + base_revision = gate(base_revision, model_name, old_model_name, mapper_moved_name) + model_revision = base_revision if not is_peft else None + return model_revision, tokenizer_gate( + tokenizer_name, model_name, old_model_name, revision, model_revision, is_peft + ) + + return run + + +@pytest.mark.parametrize( + "label, old_model_name, model_name, is_peft, tokenizer_name, revision, mapper_moved_name," + " expected_model, expected_tokenizer", + [ + ("plain pinned load", "org/m", "org/m", False, None, "v2", False, "v2", "v2"), + ("remapped to a prequant mirror", + "org/m", "unsloth/m-bnb-4bit", False, None, "v2", True, None, None), + # The adapter's ref is not the base repo's, and the tokenizer follows the base. + ("PEFT, remote adapter", "org/ad", "org/base", True, None, "v2", False, None, None), + # ... unless the tokenizer is the adapter itself, which the caller did pin. + ("PEFT, adapter-hosted tokenizer", + "org/ad", "org/base", True, "org/ad", "v2", False, None, "v2"), + ("plain load, third-party tokenizer", + "org/m", "org/m", False, "other/tok", "v2", False, "v2", None), + # Naming the requested repo back does not survive the remap: the weights moved. + ("remapped, tokenizer named as the requested repo", + "org/m", "unsloth/m-bnb-4bit", False, "org/m", "v2", True, None, None), + ("no revision at all", + "org/m", "unsloth/m-bnb-4bit", False, None, None, True, None, None), + ], + ids = lambda v: v if isinstance(v, str) and " " in v else None, +) +def test_the_revision_decision_matrix( + label, old_model_name, model_name, is_peft, tokenizer_name, revision, mapper_moved_name, + expected_model, expected_tokenizer, +): + """One table for the whole contract: which repo each pin is allowed to reach.""" + run = _simulate_loader() + model_revision, tokenizer_revision = run( + old_model_name, model_name, is_peft, tokenizer_name, revision, mapper_moved_name + ) + assert model_revision == expected_model, label + assert tokenizer_revision == expected_tokenizer, label + + +def test_the_processor_fallback_carries_the_tokenizer_revision(): + """get_auto_processor runs when AutoProcessor raises, so it is a real load path: an + unpinned one there hands back a default-branch processor beside pinned weights.""" + function = _function(_tree(VISION), "from_pretrained", "FastBaseModel") + fallbacks = _calls(function, "get_auto_processor") + assert fallbacks, "the processor fallback must still exist" + for call in fallbacks: + keyword = _revision_kwarg(call) + assert keyword is not None, "the fallback needs the revision too" + assert getattr(keyword.value, "id", None) == "_tokenizer_revision" + + +def test_the_fp8_quantizer_takes_the_requested_revision(): + """Its output path replaces model_name, so the gate downstream drops the pin. If it + did not quantize the pinned ref itself, that ref never reaches the weights at all.""" + function = _function(_tree(LOADER_UTILS), "_offline_quantize_to_fp8") + assert "revision" in _params(function) + for callee in ("from_pretrained",): + loads = _calls(function, callee) + assert loads + for call in loads: + assert _revision_kwarg(call) is not None, ast.unparse(call.func) + + +def test_the_fp8_cache_name_is_revision_specific(): + """A shared temp dir keyed only on the repo name would serve one ref's artifact to + another, and the artifact outlives the process that built it.""" + function = _function(_tree(LOADER_UTILS), "_offline_quantize_to_fp8") + writes = [ + n + for n in ast.walk(function) + if isinstance(n, ast.AugAssign) and getattr(n.target, "id", None) == "cache_name" + ] + assert any("revision" in ast.unparse(n) for n in writes), ( + "two revisions of one repo would share a cache entry" + ) + + +def test_both_loaders_hand_the_fp8_quantizer_the_revision(): + tree = _tree(LOADER) + for class_name in ("FastLanguageModel", "FastModel"): + function = _function(tree, "from_pretrained", class_name) + calls = _calls(function, "_offline_quantize_to_fp8") + assert calls, f"{class_name} must still quantize on the fly" + for call in calls: + keyword = _revision_kwarg(call) + assert keyword is not None + assert getattr(keyword.value, "id", None) == "revision", ( + "the fp8 source is still the caller's own repo here" + ) + + +def test_a_pinned_config_is_not_handed_to_the_vllm_path(): + """FastBaseModel skips its config load while auto_config is set, so passing one read + at the pinned ref would pair it with the default-branch weights vLLM fetches.""" + function = _function(_tree(LOADER), "from_pretrained", "FastModel") + dispatches = [ + c + for c in _calls(function, "from_pretrained") + if ast.unparse(c.func).startswith("FastBaseModel") + ] + assert dispatches + for call in dispatches: + keyword = next((k for k in call.keywords if k.arg == "auto_config"), None) + assert keyword is not None + name = getattr(keyword.value, "id", None) + assert name is not None and name != "model_config", ( + "the probed config must go through the vLLM gate first" + ) + guards = [ + n + for n in ast.walk(function) + if isinstance(n, ast.If) + and any( + isinstance(b, ast.Assign) + and any(getattr(t, "id", None) == name for t in b.targets) + and isinstance(b.value, ast.Constant) + and b.value.value is None + for b in n.body + ) + ] + assert guards, f"{name} is never withheld" + test = ast.unparse(guards[0].test) + for token in ("fast_inference", "is_vLLM_available", "user_config", "revision"): + assert token in test, token diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index 0f898cf99f..aead8d8211 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -175,16 +175,22 @@ def _revision_for_resolved_repo( def _revision_for_tokenizer_repo( - tokenizer_name, model_name, old_model_name, revision, model_revision + tokenizer_name, model_name, old_model_name, revision, model_revision, is_peft = False ): """Pick the revision for whichever repo the tokenizer is actually read from. - It is not always the base model's: an adapter-hosted tokenizer keeps the caller's ref, - while an unset tokenizer_name follows the resolved model_name and so takes whatever the - model load itself uses (None on a PEFT load, whose ref belongs to the adapter). + It is not always the base model's: an adapter-hosted tokenizer is a separate repo with + its own history, so it keeps the caller's ref even though the base model does not. An + unset tokenizer_name follows the resolved model_name and so takes whatever the model + load itself uses (None on a PEFT load, whose ref belongs to the adapter). + + On a plain load the tokenizer belongs to the same model as the weights, so it follows + model_revision even when the caller named its repo directly: a remap has already dropped + the pin off the weights, and a pinned tokenizer beside a mirror's default-branch weights + is the ref mismatch this whole gate exists to avoid. """ repo = tokenizer_name if tokenizer_name else model_name - if repo == old_model_name: + if is_peft and repo == old_model_name: return revision if repo == model_name: return model_revision @@ -571,7 +577,10 @@ def from_pretrained( load_in_8bit, load_in_16bit, ) - model_name = _offline_quantize_to_fp8(model_name, fp8_mode, text_only = text_only) + # Still the caller's repo here, so their ref is the one to quantize from. + model_name = _offline_quantize_to_fp8( + model_name, fp8_mode, text_only = text_only, revision = revision + ) else: assert new_model_name is not None model_name = new_model_name @@ -937,7 +946,7 @@ def from_pretrained( trust_remote_code = trust_remote_code, revision = model_revision, tokenizer_revision = _revision_for_tokenizer_repo( - tokenizer_name, model_name, old_model_name, revision, model_revision + tokenizer_name, model_name, old_model_name, revision, model_revision, is_peft ), fast_inference = fast_inference, gpu_memory_utilization = gpu_memory_utilization, @@ -1309,7 +1318,10 @@ def from_pretrained( load_in_8bit, load_in_16bit, ) - model_name = _offline_quantize_to_fp8(model_name, fp8_mode, text_only = text_only) + # Still the caller's repo here, so their ref is the one to quantize from. + model_name = _offline_quantize_to_fp8( + model_name, fp8_mode, text_only = text_only, revision = revision + ) else: assert new_model_name is not None model_name = new_model_name @@ -1877,6 +1889,20 @@ def _dispatch_diffusion(): # On a PEFT load model_name is the base model, which the caller's ref is not for. model_revision = base_revision if not is_peft else None + # vLLM takes no revision and fetches the default branch, so FastBaseModel drops the + # pin. The config probed above was read at that pin, and handing it down would skip + # the reload and pair a pinned config with default-branch weights. A config the + # caller passed in is theirs either way, so only withhold one we read ourselves. + dispatch_config = model_config + if ( + dispatch_config is not None + and user_config is None + and model_revision is not None + and fast_inference + and is_vLLM_available() + ): + dispatch_config = None + model, tokenizer = FastBaseModel.from_pretrained( model_name = model_name, max_seq_length = max_seq_length, @@ -1890,7 +1916,7 @@ def _dispatch_diffusion(): trust_remote_code = trust_remote_code, revision = model_revision, tokenizer_revision = _revision_for_tokenizer_repo( - tokenizer_name, model_name, old_model_name, revision, model_revision + tokenizer_name, model_name, old_model_name, revision, model_revision, is_peft ), model_types = model_types, tokenizer_name = tokenizer_name, @@ -1899,7 +1925,7 @@ def _dispatch_diffusion(): supports_sdpa = supports_sdpa, whisper_language = whisper_language, whisper_task = whisper_task, - auto_config = model_config, + auto_config = dispatch_config, offload_embedding = offload_embedding, float32_mixed_precision = float32_mixed_precision, # Pass vLLM/inference parameters diff --git a/unsloth/models/loader_utils.py b/unsloth/models/loader_utils.py index 68d55dc404..f63fa4d60d 100644 --- a/unsloth/models/loader_utils.py +++ b/unsloth/models/loader_utils.py @@ -314,11 +314,16 @@ def _offline_quantize_to_fp8( fp8_mode: str, *, text_only: bool = False, + revision: str = None, ) -> str: """Quantize the model to fp8 via torchao, save to a temp dir, return its path. For vllm >= 0.12.0, prefer dynamic quantization in vllm instead (via hf_overrides={"quantization_config_file": "torchao_config.json"}). + + The caller's revision has to reach the source loads, and the cache name has to name it + too: the returned path replaces model_name, so the revision gate downstream drops the + pin, and two refs of one repo would otherwise share (and reuse) a single artifact. """ from transformers import ( AutoModelForCausalLM, @@ -329,7 +334,7 @@ def _offline_quantize_to_fp8( AutoConfig, ) - config = AutoConfig.from_pretrained(model_name) + config = AutoConfig.from_pretrained(model_name, revision = revision) is_vlm = any( x.endswith(("ForConditionalGeneration", "ForVisionText2Text")) for x in (getattr(config, "architectures", None) or []) @@ -356,6 +361,9 @@ def _offline_quantize_to_fp8( temp_dir = tempfile.gettempdir() # Cache text-only and full-VLM artifacts separately so neither reuses the other. #5816 cache_name = model_name.split("/")[-1] + "-fp8-" + fp8_mode + if revision is not None: + # Slashes and dots would escape the temp dir; a branch name may hold both. + cache_name += "-rev-" + re.sub(r"[^0-9A-Za-z_-]", "_", revision) if text_config is not None: cache_name += "-text-only" new_model_name = os.path.join(temp_dir, cache_name) @@ -375,9 +383,10 @@ def _offline_quantize_to_fp8( model = auto_model.from_pretrained( model_name, config = config, + revision = revision, **load_kwargs, ) - tokenizer = auto_processor.from_pretrained(model_name) + tokenizer = auto_processor.from_pretrained(model_name, revision = revision) model.save_pretrained(new_model_name, safe_serialization = False) del model for _ in range(2): diff --git a/unsloth/models/vision.py b/unsloth/models/vision.py index ca67259c90..d7efa55d84 100644 --- a/unsloth/models/vision.py +++ b/unsloth/models/vision.py @@ -1544,6 +1544,7 @@ def _acquire_processor(lfo): trust_remote_code = trust_remote_code, cache_dir = kwargs.get("cache_dir"), local_files_only = lfo, + revision = _tokenizer_revision, ) except Exception: # Swallow so the manual fallback / entry-point retry can run. From 129061b3c69c6e10505f7b03b1493f4a53f80a77 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Sun, 2 Aug 2026 16:43:52 +0000 Subject: [PATCH 15/20] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- tests/python/test_revision_forwarding.py | 87 ++++++++++++++++++------ unsloth/models/loader.py | 7 +- 2 files changed, 71 insertions(+), 23 deletions(-) diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py index 0ae9cb3dfb..79f5f31a96 100644 --- a/tests/python/test_revision_forwarding.py +++ b/tests/python/test_revision_forwarding.py @@ -545,7 +545,8 @@ def _simulate_loader(): tree = ast.parse(LOADER.read_text(encoding = "utf-8")) namespace = {"logger": types.SimpleNamespace(warning_once = lambda *a, **k: None)} functions = [ - n for n in tree.body + n + for n in tree.body if isinstance(n, ast.FunctionDef) and n.name in ("_revision_for_resolved_repo", "_revision_for_tokenizer_repo") ] @@ -572,26 +573,68 @@ def run(old_model_name, model_name, is_peft, tokenizer_name, revision, mapper_mo " expected_model, expected_tokenizer", [ ("plain pinned load", "org/m", "org/m", False, None, "v2", False, "v2", "v2"), - ("remapped to a prequant mirror", - "org/m", "unsloth/m-bnb-4bit", False, None, "v2", True, None, None), + ( + "remapped to a prequant mirror", + "org/m", + "unsloth/m-bnb-4bit", + False, + None, + "v2", + True, + None, + None, + ), # The adapter's ref is not the base repo's, and the tokenizer follows the base. ("PEFT, remote adapter", "org/ad", "org/base", True, None, "v2", False, None, None), # ... unless the tokenizer is the adapter itself, which the caller did pin. - ("PEFT, adapter-hosted tokenizer", - "org/ad", "org/base", True, "org/ad", "v2", False, None, "v2"), - ("plain load, third-party tokenizer", - "org/m", "org/m", False, "other/tok", "v2", False, "v2", None), + ( + "PEFT, adapter-hosted tokenizer", + "org/ad", + "org/base", + True, + "org/ad", + "v2", + False, + None, + "v2", + ), + ( + "plain load, third-party tokenizer", + "org/m", + "org/m", + False, + "other/tok", + "v2", + False, + "v2", + None, + ), # Naming the requested repo back does not survive the remap: the weights moved. - ("remapped, tokenizer named as the requested repo", - "org/m", "unsloth/m-bnb-4bit", False, "org/m", "v2", True, None, None), - ("no revision at all", - "org/m", "unsloth/m-bnb-4bit", False, None, None, True, None, None), + ( + "remapped, tokenizer named as the requested repo", + "org/m", + "unsloth/m-bnb-4bit", + False, + "org/m", + "v2", + True, + None, + None, + ), + ("no revision at all", "org/m", "unsloth/m-bnb-4bit", False, None, None, True, None, None), ], ids = lambda v: v if isinstance(v, str) and " " in v else None, ) def test_the_revision_decision_matrix( - label, old_model_name, model_name, is_peft, tokenizer_name, revision, mapper_moved_name, - expected_model, expected_tokenizer, + label, + old_model_name, + model_name, + is_peft, + tokenizer_name, + revision, + mapper_moved_name, + expected_model, + expected_tokenizer, ): """One table for the whole contract: which repo each pin is allowed to reach.""" run = _simulate_loader() @@ -635,9 +678,9 @@ def test_the_fp8_cache_name_is_revision_specific(): for n in ast.walk(function) if isinstance(n, ast.AugAssign) and getattr(n.target, "id", None) == "cache_name" ] - assert any("revision" in ast.unparse(n) for n in writes), ( - "two revisions of one repo would share a cache entry" - ) + assert any( + "revision" in ast.unparse(n) for n in writes + ), "two revisions of one repo would share a cache entry" def test_both_loaders_hand_the_fp8_quantizer_the_revision(): @@ -649,9 +692,9 @@ def test_both_loaders_hand_the_fp8_quantizer_the_revision(): for call in calls: keyword = _revision_kwarg(call) assert keyword is not None - assert getattr(keyword.value, "id", None) == "revision", ( - "the fp8 source is still the caller's own repo here" - ) + assert ( + getattr(keyword.value, "id", None) == "revision" + ), "the fp8 source is still the caller's own repo here" def test_a_pinned_config_is_not_handed_to_the_vllm_path(): @@ -668,9 +711,9 @@ def test_a_pinned_config_is_not_handed_to_the_vllm_path(): keyword = next((k for k in call.keywords if k.arg == "auto_config"), None) assert keyword is not None name = getattr(keyword.value, "id", None) - assert name is not None and name != "model_config", ( - "the probed config must go through the vLLM gate first" - ) + assert ( + name is not None and name != "model_config" + ), "the probed config must go through the vLLM gate first" guards = [ n for n in ast.walk(function) diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index aead8d8211..d6c7a56581 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -175,7 +175,12 @@ def _revision_for_resolved_repo( def _revision_for_tokenizer_repo( - tokenizer_name, model_name, old_model_name, revision, model_revision, is_peft = False + tokenizer_name, + model_name, + old_model_name, + revision, + model_revision, + is_peft = False, ): """Pick the revision for whichever repo the tokenizer is actually read from. From 05046e0df9d95bc8659053e99053a85ebd148f22 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 2 Aug 2026 17:13:42 +0000 Subject: [PATCH 16/20] Drop the vLLM pin before the probe, and key fp8 artifacts on the raw ref FastModel withheld the probed config from FastBaseModel on the vLLM path, but model_types, auto_model and the text-only decision had already been derived from it, so default-branch weights could load with pinned-ref dispatch. The drop now happens before the probe instead, using the same predicate FastBaseModel does, which makes that guard a no-op on this path and lets the config go down untouched again. FastLanguageModel keeps its drop inside llama.py: that one also turns fast_inference off on pre-Volta GPUs and for a num_labels load, and the loader cannot see either without duplicating the device checks, so gating early there would discard a pin llama.py would have honoured. The fp8 cache name sanitized the ref by replacing every unsafe character with the same one, so release/v1 and release.v1 shared a directory and the second load reused the first ref's artifact. A digest of the raw ref now rides along with the readable form. --- tests/python/test_revision_forwarding.py | 79 ++++++++++++++++-------- unsloth/models/loader.py | 28 ++++----- unsloth/models/loader_utils.py | 9 ++- 3 files changed, 73 insertions(+), 43 deletions(-) diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py index 79f5f31a96..e90d5b619b 100644 --- a/tests/python/test_revision_forwarding.py +++ b/tests/python/test_revision_forwarding.py @@ -678,9 +678,15 @@ def test_the_fp8_cache_name_is_revision_specific(): for n in ast.walk(function) if isinstance(n, ast.AugAssign) and getattr(n.target, "id", None) == "cache_name" ] - assert any( - "revision" in ast.unparse(n) for n in writes - ), "two revisions of one repo would share a cache entry" + assert writes + guarded = [ + n + for n in ast.walk(function) + if isinstance(n, ast.If) + and "revision" in ast.unparse(n.test) + and any(n.lineno <= w.lineno <= n.end_lineno for w in writes) + ] + assert guarded, "two revisions of one repo would share a cache entry" def test_both_loaders_hand_the_fp8_quantizer_the_revision(): @@ -697,10 +703,35 @@ def test_both_loaders_hand_the_fp8_quantizer_the_revision(): ), "the fp8 source is still the caller's own repo here" -def test_a_pinned_config_is_not_handed_to_the_vllm_path(): - """FastBaseModel skips its config load while auto_config is set, so passing one read - at the pinned ref would pair it with the default-branch weights vLLM fetches.""" +def test_the_vllm_drop_happens_before_the_config_probe(): + """model_types, auto_model and the text-only decision all come off the probed config. + Reading it at a ref vLLM will not fetch picks the dispatch for a different model, so + the pin has to be gone before the probe, not just before the dispatch.""" function = _function(_tree(LOADER), "from_pretrained", "FastModel") + drops = [ + n + for n in ast.walk(function) + if isinstance(n, ast.If) + and "is_vLLM_available" in ast.unparse(n.test) + and any( + isinstance(b, ast.Assign) + and any(getattr(t, "id", None) == "base_revision" for t in b.targets) + and isinstance(b.value, ast.Constant) + and b.value.value is None + for b in n.body + ) + ] + assert drops, "FastModel never drops base_revision for the vLLM path" + probes = [ + c + for c in _calls(function, "from_pretrained") + if ast.unparse(c.func).split(".")[0] in ("AutoConfig", "PeftConfig") + ] + assert probes + assert drops[0].end_lineno < min(c.lineno for c in probes), ( + "the probe would read a ref the weights will not be at" + ) + # The probed config goes down untouched again, so nothing may re-gate it at dispatch. dispatches = [ c for c in _calls(function, "from_pretrained") @@ -710,23 +741,19 @@ def test_a_pinned_config_is_not_handed_to_the_vllm_path(): for call in dispatches: keyword = next((k for k in call.keywords if k.arg == "auto_config"), None) assert keyword is not None - name = getattr(keyword.value, "id", None) - assert ( - name is not None and name != "model_config" - ), "the probed config must go through the vLLM gate first" - guards = [ - n - for n in ast.walk(function) - if isinstance(n, ast.If) - and any( - isinstance(b, ast.Assign) - and any(getattr(t, "id", None) == name for t in b.targets) - and isinstance(b.value, ast.Constant) - and b.value.value is None - for b in n.body - ) - ] - assert guards, f"{name} is never withheld" - test = ast.unparse(guards[0].test) - for token in ("fast_inference", "is_vLLM_available", "user_config", "revision"): - assert token in test, token + assert getattr(keyword.value, "id", None) == "model_config" + + +def test_the_fp8_cache_key_survives_a_lossy_sanitization(): + """The readable half replaces every unsafe character with the same one, so `a/b` and + `a.b` collapse together. Only a digest of the raw ref keeps them apart.""" + function = _function(_tree(LOADER_UTILS), "_offline_quantize_to_fp8") + source = ast.unparse(function) + assert "sha256" in source or "blake2" in source, "the sanitized name alone collides" + digests = [ + n + for n in ast.walk(function) + if isinstance(n, ast.Call) and "sha256" in ast.unparse(n.func) + ] + assert digests + assert any("revision" in ast.unparse(n) for n in digests), "hash the ref, not the repo" diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index d6c7a56581..01580b386d 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -1362,6 +1362,18 @@ def from_pretrained( base_revision = _revision_for_resolved_repo( revision, model_name, old_model_name, mapper_moved_name ) + # vLLM takes no revision and fetches the default branch, so this pin is already dead + # for the weights. Drop it here rather than at the dispatch: model_types, auto_model + # and the text-only decision all come off the config probed below, and reading that + # at a ref the weights will not be at picks the dispatch for the wrong model. Same + # predicate FastBaseModel uses, so its own guard is a no-op on this path. An adapter + # still keeps `revision`: peft loads it in-process, not through vLLM. + if base_revision is not None and fast_inference and is_vLLM_available(): + logger.warning_once( + f"Unsloth: Ignoring revision = `{base_revision}` since vLLM loads weights " + "from the default branch. Use `fast_inference = False` to load a pinned revision." + ) + base_revision = None # First check if it's a normal model via AutoConfig from huggingface_hub.utils import ( @@ -1894,20 +1906,6 @@ def _dispatch_diffusion(): # On a PEFT load model_name is the base model, which the caller's ref is not for. model_revision = base_revision if not is_peft else None - # vLLM takes no revision and fetches the default branch, so FastBaseModel drops the - # pin. The config probed above was read at that pin, and handing it down would skip - # the reload and pair a pinned config with default-branch weights. A config the - # caller passed in is theirs either way, so only withhold one we read ourselves. - dispatch_config = model_config - if ( - dispatch_config is not None - and user_config is None - and model_revision is not None - and fast_inference - and is_vLLM_available() - ): - dispatch_config = None - model, tokenizer = FastBaseModel.from_pretrained( model_name = model_name, max_seq_length = max_seq_length, @@ -1930,7 +1928,7 @@ def _dispatch_diffusion(): supports_sdpa = supports_sdpa, whisper_language = whisper_language, whisper_task = whisper_task, - auto_config = dispatch_config, + auto_config = model_config, offload_embedding = offload_embedding, float32_mixed_precision = float32_mixed_precision, # Pass vLLM/inference parameters diff --git a/unsloth/models/loader_utils.py b/unsloth/models/loader_utils.py index f63fa4d60d..73b97929ea 100644 --- a/unsloth/models/loader_utils.py +++ b/unsloth/models/loader_utils.py @@ -13,6 +13,7 @@ # limitations under the License. from ..device_type import DEVICE_TYPE_TORCH +import hashlib import importlib import os import torch @@ -362,8 +363,12 @@ def _offline_quantize_to_fp8( # Cache text-only and full-VLM artifacts separately so neither reuses the other. #5816 cache_name = model_name.split("/")[-1] + "-fp8-" + fp8_mode if revision is not None: - # Slashes and dots would escape the temp dir; a branch name may hold both. - cache_name += "-rev-" + re.sub(r"[^0-9A-Za-z_-]", "_", revision) + # Slashes and dots would escape the temp dir, so the readable half is sanitized and + # therefore lossy: `release/v1` and `release.v1` collapse to one name. A digest of + # the raw ref rides along so two refs never share (and silently reuse) an artifact. + digest = hashlib.sha256(revision.encode("utf-8")).hexdigest()[:12] + readable = re.sub(r"[^0-9A-Za-z_-]", "_", revision)[:40] + cache_name += "-rev-" + readable + "-" + digest if text_config is not None: cache_name += "-text-only" new_model_name = os.path.join(temp_dir, cache_name) From b35d06a51fc38f275f346eb3b4028055bbb0af3c Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Sun, 2 Aug 2026 17:14:37 +0000 Subject: [PATCH 17/20] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- tests/python/test_revision_forwarding.py | 10 ++++------ 1 file changed, 4 insertions(+), 6 deletions(-) diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py index e90d5b619b..df0a170535 100644 --- a/tests/python/test_revision_forwarding.py +++ b/tests/python/test_revision_forwarding.py @@ -728,9 +728,9 @@ def test_the_vllm_drop_happens_before_the_config_probe(): if ast.unparse(c.func).split(".")[0] in ("AutoConfig", "PeftConfig") ] assert probes - assert drops[0].end_lineno < min(c.lineno for c in probes), ( - "the probe would read a ref the weights will not be at" - ) + assert drops[0].end_lineno < min( + c.lineno for c in probes + ), "the probe would read a ref the weights will not be at" # The probed config goes down untouched again, so nothing may re-gate it at dispatch. dispatches = [ c @@ -751,9 +751,7 @@ def test_the_fp8_cache_key_survives_a_lossy_sanitization(): source = ast.unparse(function) assert "sha256" in source or "blake2" in source, "the sanitized name alone collides" digests = [ - n - for n in ast.walk(function) - if isinstance(n, ast.Call) and "sha256" in ast.unparse(n.func) + n for n in ast.walk(function) if isinstance(n, ast.Call) and "sha256" in ast.unparse(n.func) ] assert digests assert any("revision" in ast.unparse(n) for n in digests), "hash the ref, not the repo" From 738116a9a864039c624eeaa71f3f283800039750 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 2 Aug 2026 17:48:19 +0000 Subject: [PATCH 18/20] Gate the language probe on vLLM too, spare the adapter probe, stamp the saved ref FastLanguageModel still probed the config at the pinned ref while llama.py dropped that same ref for its vLLM load, so model_types could pick the architecture class off one ref and load weights from another. It now drops the pin before the probe like FastModel does, through _vllm_will_load_weights in llama.py, which llama.py itself now calls: the language path also falls back in-process on pre-Volta GPUs and for a num_labels load, so the predicate has to live where those checks are rather than be guessed at by the loader. That drop runs before is_peft is known, and it was zeroing the ref the PeftConfig probe reads. An adapter is loaded in-process by peft, so it keeps the ref: adapter_revision holds the value from before the vLLM drop. Pinning the tokenizer also desynced the save path, which restores tokenizer.model from tokenizer.name_or_path and so had no idea which branch to read. The loaded ref is now stamped on the tokenizer the way local_files_only and cache_dir already are, and the sentencepiece probe, its memo key and the restore all use it. --- tests/python/test_revision_forwarding.py | 104 ++++++++++++++++++++++- unsloth/models/llama.py | 23 ++++- unsloth/models/loader.py | 28 ++++-- unsloth/models/loader_utils.py | 33 +++++++ unsloth/save.py | 19 +++-- unsloth/tokenizer_utils.py | 6 ++ 6 files changed, 200 insertions(+), 13 deletions(-) diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py index df0a170535..4181fd9a92 100644 --- a/tests/python/test_revision_forwarding.py +++ b/tests/python/test_revision_forwarding.py @@ -21,6 +21,7 @@ VISION = REPO / "unsloth" / "models" / "vision.py" TOKENIZER_UTILS = REPO / "unsloth" / "tokenizer_utils.py" LOADER_UTILS = REPO / "unsloth" / "models" / "loader_utils.py" +SAVE = REPO / "unsloth" / "save.py" def _tree(path): @@ -262,8 +263,11 @@ def test_both_loader_paths_gate_before_and_after_resolution(): keyword = _revision_kwarg(probe) if keyword is None: continue # the PEFT base-model probe deliberately pins nothing - assert ( - getattr(keyword.value, "id", None) == "base_revision" + # adapter_revision is the same gated value, taken before the vLLM drop that + # only the base model's config and weights answer to. + assert getattr(keyword.value, "id", None) in ( + "base_revision", + "adapter_revision", ), f"probe at line {probe.lineno} uses the ungated revision" gated += 1 assert gated >= 2, f"{class_name} must gate its AutoConfig and PeftConfig probes" @@ -755,3 +759,99 @@ def test_the_fp8_cache_key_survives_a_lossy_sanitization(): ] assert digests assert any("revision" in ast.unparse(n) for n in digests), "hash the ref, not the repo" + + +def test_the_peft_probe_keeps_the_adapter_ref_under_vllm(): + """The vLLM drop runs before is_peft is known. An adapter is loaded in-process by peft, + so zeroing its probe would read the default branch and either miss PEFT entirely or + resolve a different base model before attaching the pinned adapter.""" + tree = _tree(LOADER) + for class_name in ("FastLanguageModel", "FastModel"): + function = _function(tree, "from_pretrained", class_name) + probes = [ + c for c in _calls(function, "from_pretrained") if ast.unparse(c.func).startswith("PeftConfig") + ] + assert probes, f"{class_name} must still probe for an adapter" + for call in probes: + keyword = _revision_kwarg(call) + assert keyword is not None + assert getattr(keyword.value, "id", None) == "adapter_revision", ( + "the adapter probe must not take the base model's gated ref" + ) + + +def test_both_loaders_drop_the_vllm_pin_before_the_probe(): + """model_types picks the architecture class off the probed config, so reading it at a + ref vLLM will not fetch dispatches the wrong one.""" + tree = _tree(LOADER) + for class_name in ("FastLanguageModel", "FastModel"): + function = _function(tree, "from_pretrained", class_name) + drops = [ + n + for n in ast.walk(function) + if isinstance(n, ast.If) + and any( + isinstance(b, ast.Assign) + and any(getattr(t, "id", None) == "base_revision" for t in b.targets) + and isinstance(b.value, ast.Constant) + and b.value.value is None + for b in n.body + ) + ] + assert drops, f"{class_name} never drops base_revision for the vLLM path" + probes = [ + c + for c in _calls(function, "from_pretrained") + if ast.unparse(c.func).split(".")[0] in ("AutoConfig", "PeftConfig") + ] + assert probes + assert drops[0].end_lineno < min(c.lineno for c in probes), ( + f"{class_name} probes at a ref the weights will not be at" + ) + + +def test_llama_owns_the_vllm_predicate_the_loader_gates_on(): + """FastLanguageModel also falls back in-process on pre-Volta GPUs and for a num_labels + load, so the loader cannot gate on `fast_inference and is_vLLM_available()` the way the + FastModel path does. One helper, used by both, or the two drift apart.""" + tree = _tree(LLAMA) + helper = _function(tree, "_vllm_will_load_weights") + source = ast.unparse(helper) + for token in ("is_vLLM_available", "get_device_capability", "hip", "num_labels"): + assert token in source, token + + guard = _function(tree, "from_pretrained", "FastLlamaModel") + assert _calls(guard, "_vllm_will_load_weights"), "llama.py must use its own helper" + loader = _function(_tree(LOADER), "from_pretrained", "FastLanguageModel") + assert _calls(loader, "_vllm_will_load_weights"), "the loader must gate on the same one" + + +def test_a_pinned_tokenizer_is_stamped_for_the_save_path(): + """save.py restores tokenizer.model from tokenizer.name_or_path, which names the repo + but not the branch, so a merged export would copy the default branch's asset.""" + stamps = _calls(_function(_tree(TOKENIZER_UTILS), "load_correct_tokenizer"), "_mark_loaded_revision") + assert stamps, "the loaded ref has to travel with the tokenizer" + assert any( + any(getattr(a, "id", None) == "revision" for a in c.args) for c in stamps + ), "stamp the ref that was actually loaded" + + tree = _tree(LOADER_UTILS) + assert _function(tree, "_mark_loaded_revision") + assert _function(tree, "_tokenizer_revision") + assert "revision" in _params(_function(tree, "_resolve_hub_repo_cached_file")) + + +@pytest.mark.parametrize("callee", ["_resolve_hub_repo_cached_file", "hf_hub_download", "model_info"]) +def test_the_sentencepiece_restore_reads_the_stamped_ref(callee): + tree = _tree(SAVE) + functions = [ + n + for n in ast.walk(tree) + if isinstance(n, ast.FunctionDef) + and n.name in ("_has_tokenizer_model", "_preserve_sentencepiece_tokenizer_assets") + ] + assert functions + calls = [c for f in functions for c in _calls(f, callee)] + assert calls, f"{callee} not found on the save path" + for call in calls: + assert _revision_kwarg(call) is not None, f"{callee} at line {call.lineno} drops the ref" diff --git a/unsloth/models/llama.py b/unsloth/models/llama.py index cc0965d54a..d4688d30a1 100644 --- a/unsloth/models/llama.py +++ b/unsloth/models/llama.py @@ -2244,6 +2244,26 @@ def unsloth_fast_generate(self, *args, **kwargs): return output +def _vllm_will_load_weights(fast_inference, num_labels = None): + """Whether vLLM, which takes no revision, ends up owning the weight load. + + The loader has to answer this before it probes the config, since that probe's ref + decides which architecture class the load is dispatched to. Mirrors the checks at the + top of from_pretrained below, which calls this too so the two cannot drift. + """ + if not fast_inference or num_labels is not None: + return False + # from_pretrained clears fast_inference when vLLM is missing and then re-enables it on + # hip, so hip ends up True either way. + if DEVICE_TYPE == "hip": + return True + if not is_vLLM_available(): + return False + if DEVICE_TYPE == "cuda" and torch.cuda.get_device_capability()[0] < 7: + return False + return True + + class FastLlamaModel: @staticmethod def _prepare_for_qat(model, qat_scheme): @@ -2342,7 +2362,8 @@ def from_pretrained( # Only vLLM cannot take a revision. fast_inference may have just been turned # off above, and a num_labels load goes in-process regardless; both of those # can honour the pin, so use the same predicate as the prefetch warm below. - if fast_inference and num_labels is None and revision is not None: + # Through the helper, which the loader also uses to gate its config probe. + if _vllm_will_load_weights(fast_inference, num_labels) and revision is not None: # load_vllm takes no revision, so vLLM fetches the default branch. Pinning # only the config and tokenizer would mix two refs in one model. logger.warning_once( diff --git a/unsloth/models/loader.py b/unsloth/models/loader.py index 01580b386d..533e74681e 100644 --- a/unsloth/models/loader.py +++ b/unsloth/models/loader.py @@ -27,7 +27,7 @@ DISABLE_SDPA_MODEL_NAMES, ) from .granite import FastGraniteModel -from .llama import FastLlamaModel, logger +from .llama import FastLlamaModel, logger, _vllm_will_load_weights from .mistral import FastMistralModel from .qwen2 import FastQwen2Model from .qwen3 import FastQwen3Model @@ -620,6 +620,22 @@ def from_pretrained( base_revision = _revision_for_resolved_repo( revision, model_name, old_model_name, mapper_moved_name ) + # The PeftConfig probe below reads the adapter repo, which peft loads in-process, so + # it keeps the ref even when vLLM takes the base model's away just after. + adapter_revision = base_revision + # vLLM takes no revision and fetches the default branch, so this pin is already dead + # for the weights. Drop it before the probe: model_types picks the architecture class + # off that config, and reading it at a ref the weights will not be at dispatches the + # wrong one. The predicate lives in llama.py, which also falls back in-process on + # pre-Volta GPUs and for a num_labels load; both of those can still honour the pin. + if base_revision is not None and _vllm_will_load_weights( + fast_inference, kwargs.get("num_labels") + ): + logger.warning_once( + f"Unsloth: Ignoring revision = `{base_revision}` since vLLM loads weights " + "from the default branch. Use `fast_inference = False` to load a pinned revision." + ) + base_revision = None # First check if it's a normal model via AutoConfig from huggingface_hub.utils import ( @@ -670,7 +686,7 @@ def from_pretrained( peft_config = PeftConfig.from_pretrained( model_name, token = token, - revision = base_revision, + revision = adapter_revision, trust_remote_code = trust_remote_code, local_files_only = local_files_only, ) @@ -1362,12 +1378,14 @@ def from_pretrained( base_revision = _revision_for_resolved_repo( revision, model_name, old_model_name, mapper_moved_name ) + # The PeftConfig probe below reads the adapter repo, which peft loads in-process, so + # it keeps the ref even when vLLM takes the base model's away just after. + adapter_revision = base_revision # vLLM takes no revision and fetches the default branch, so this pin is already dead # for the weights. Drop it here rather than at the dispatch: model_types, auto_model # and the text-only decision all come off the config probed below, and reading that # at a ref the weights will not be at picks the dispatch for the wrong model. Same - # predicate FastBaseModel uses, so its own guard is a no-op on this path. An adapter - # still keeps `revision`: peft loads it in-process, not through vLLM. + # predicate FastBaseModel uses, so its own guard is a no-op on this path. if base_revision is not None and fast_inference and is_vLLM_available(): logger.warning_once( f"Unsloth: Ignoring revision = `{base_revision}` since vLLM loads weights " @@ -1451,7 +1469,7 @@ def _dispatch_diffusion(): peft_config = PeftConfig.from_pretrained( model_name, token = token, - revision = base_revision, + revision = adapter_revision, trust_remote_code = trust_remote_code, local_files_only = local_files_only, ) diff --git a/unsloth/models/loader_utils.py b/unsloth/models/loader_utils.py index 73b97929ea..8ea3ed8ad4 100644 --- a/unsloth/models/loader_utils.py +++ b/unsloth/models/loader_utils.py @@ -898,6 +898,37 @@ def _get_effective_local_files_only(kwargs): # The load's cache_dir travels with it too: saving derives one from HF_HUB_CACHE / # HF_HOME, which does not see a caller-supplied cache. _LOADED_CACHE_DIR_ATTR = "_unsloth_loaded_cache_dir" +# So does the ref it was read at. Saving restores sentencepiece assets from +# tokenizer.name_or_path, which names the repo but not the branch, so without this stamp a +# merged export copies the default branch's tokenizer.model beside pinned metadata, or +# misses the file when it only exists on the pinned ref. +_LOADED_REVISION_ATTR = "_unsloth_loaded_revision" + + +def _mark_loaded_revision(result, revision): + """Stamp the ref a tokenizer/processor was loaded at onto the returned objects.""" + if revision is None: + return result + for obj in result if isinstance(result, (tuple, list)) else (result,): + try: + targets = (obj, getattr(obj, "tokenizer", None)) + except Exception: + targets = (obj,) + for target in targets: + if target is None: + continue + # Objects that reject new attributes (__slots__) are skipped. + try: + setattr(target, _LOADED_REVISION_ATTR, str(revision)) + except Exception: + pass + return result + + +def _tokenizer_revision(tokenizer): + """The ref this tokenizer was loaded at, or None for the default branch.""" + tokenizer = tokenizer.tokenizer if hasattr(tokenizer, "tokenizer") else tokenizer + return getattr(tokenizer, _LOADED_REVISION_ATTR, None) def _mark_loaded_local_files_only(result, cache_dir = None): @@ -1280,6 +1311,7 @@ def _resolve_hub_repo_cached_file( token = None, cache_dir = None, local_files_only = True, + revision = None, ): """Return a cached file path under a Hub snapshot, or None if absent.""" local_dir = _resolve_hub_repo_local_dir( @@ -1287,6 +1319,7 @@ def _resolve_hub_repo_cached_file( token = token, cache_dir = cache_dir, local_files_only = local_files_only, + revision = revision, filenames = (filename,), ) if local_dir is None: diff --git a/unsloth/save.py b/unsloth/save.py index dd0fceb235..2113d8e20c 100644 --- a/unsloth/save.py +++ b/unsloth/save.py @@ -68,6 +68,7 @@ class Peft_Linear4bit: get_model_name, _resolve_hub_repo_cached_file, _tokenizer_cache_dir, + _tokenizer_revision, _tokenizer_wants_local_only, ) from .models._utils import _convert_torchao_model @@ -460,8 +461,11 @@ def _has_tokenizer_model(tokenizer, token = None): return False if os.path.isdir(source): return os.path.isfile(os.path.join(source, "tokenizer.model")) - if source in _TOKENIZER_MODEL_CACHE: - return _TOKENIZER_MODEL_CACHE[source] + # Refs of one repo can differ in whether they ship the asset, so memoize per ref. + revision = _tokenizer_revision(tokenizer) + cache_key = (source, revision) + if cache_key in _TOKENIZER_MODEL_CACHE: + return _TOKENIZER_MODEL_CACHE[cache_key] # Hub repo id: probe local cache before model_info (issue #7481). cache_dir = _tokenizer_cache_dir(tokenizer) or os.environ.get("HF_HUB_CACHE") @@ -476,23 +480,26 @@ def _has_tokenizer_model(tokenizer, token = None): token = token, local_files_only = True, cache_dir = cache_dir, + revision = revision, ) if cached_path is not None: - _TOKENIZER_MODEL_CACHE[source] = True + _TOKENIZER_MODEL_CACHE[cache_key] = True return True if _tokenizer_wants_local_only(tokenizer): return False try: - repo_info = HfApi(token = token).model_info(source, files_metadata = False) + repo_info = HfApi(token = token).model_info( + source, revision = revision, files_metadata = False + ) except Exception: return False has_tokenizer_model = any( sibling.rfilename == "tokenizer.model" for sibling in (repo_info.siblings or []) ) - _TOKENIZER_MODEL_CACHE[source] = has_tokenizer_model + _TOKENIZER_MODEL_CACHE[cache_key] = has_tokenizer_model return has_tokenizer_model @@ -555,6 +562,7 @@ def _preserve_sentencepiece_tokenizer_assets( token = token, local_files_only = True, cache_dir = cache_dir, + revision = _tokenizer_revision(tokenizer), ) if cached_path is not None: downloaded_path = cached_path @@ -567,6 +575,7 @@ def _preserve_sentencepiece_tokenizer_assets( token = token, local_files_only = _tokenizer_wants_local_only(tokenizer), cache_dir = cache_dir, + revision = _tokenizer_revision(tokenizer), ) except Exception: downloaded_path = None diff --git a/unsloth/tokenizer_utils.py b/unsloth/tokenizer_utils.py index 46ec60c37c..9dd58484ca 100644 --- a/unsloth/tokenizer_utils.py +++ b/unsloth/tokenizer_utils.py @@ -716,6 +716,12 @@ def load_correct_tokenizer( pass tokenizer.chat_template = chat_template + # Saving restores sentencepiece assets from the repo name alone, which does not carry + # the branch this was read at, so stamp it for the save path to find. Imported here: + # models.loader_utils pulls in models._utils, which imports this module at load time. + from .models.loader_utils import _mark_loaded_revision + + _mark_loaded_revision(tokenizer, revision) return tokenizer From 40d993a43b76d1adc2003f717ec2ca2f6499ea9f Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Sun, 2 Aug 2026 17:50:01 +0000 Subject: [PATCH 19/20] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- tests/python/test_revision_forwarding.py | 24 +++++++++++++++--------- unsloth/save.py | 4 +--- 2 files changed, 16 insertions(+), 12 deletions(-) diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py index 4181fd9a92..3f529b5879 100644 --- a/tests/python/test_revision_forwarding.py +++ b/tests/python/test_revision_forwarding.py @@ -769,15 +769,17 @@ def test_the_peft_probe_keeps_the_adapter_ref_under_vllm(): for class_name in ("FastLanguageModel", "FastModel"): function = _function(tree, "from_pretrained", class_name) probes = [ - c for c in _calls(function, "from_pretrained") if ast.unparse(c.func).startswith("PeftConfig") + c + for c in _calls(function, "from_pretrained") + if ast.unparse(c.func).startswith("PeftConfig") ] assert probes, f"{class_name} must still probe for an adapter" for call in probes: keyword = _revision_kwarg(call) assert keyword is not None - assert getattr(keyword.value, "id", None) == "adapter_revision", ( - "the adapter probe must not take the base model's gated ref" - ) + assert ( + getattr(keyword.value, "id", None) == "adapter_revision" + ), "the adapter probe must not take the base model's gated ref" def test_both_loaders_drop_the_vllm_pin_before_the_probe(): @@ -805,9 +807,9 @@ def test_both_loaders_drop_the_vllm_pin_before_the_probe(): if ast.unparse(c.func).split(".")[0] in ("AutoConfig", "PeftConfig") ] assert probes - assert drops[0].end_lineno < min(c.lineno for c in probes), ( - f"{class_name} probes at a ref the weights will not be at" - ) + assert drops[0].end_lineno < min( + c.lineno for c in probes + ), f"{class_name} probes at a ref the weights will not be at" def test_llama_owns_the_vllm_predicate_the_loader_gates_on(): @@ -829,7 +831,9 @@ def test_llama_owns_the_vllm_predicate_the_loader_gates_on(): def test_a_pinned_tokenizer_is_stamped_for_the_save_path(): """save.py restores tokenizer.model from tokenizer.name_or_path, which names the repo but not the branch, so a merged export would copy the default branch's asset.""" - stamps = _calls(_function(_tree(TOKENIZER_UTILS), "load_correct_tokenizer"), "_mark_loaded_revision") + stamps = _calls( + _function(_tree(TOKENIZER_UTILS), "load_correct_tokenizer"), "_mark_loaded_revision" + ) assert stamps, "the loaded ref has to travel with the tokenizer" assert any( any(getattr(a, "id", None) == "revision" for a in c.args) for c in stamps @@ -841,7 +845,9 @@ def test_a_pinned_tokenizer_is_stamped_for_the_save_path(): assert "revision" in _params(_function(tree, "_resolve_hub_repo_cached_file")) -@pytest.mark.parametrize("callee", ["_resolve_hub_repo_cached_file", "hf_hub_download", "model_info"]) +@pytest.mark.parametrize( + "callee", ["_resolve_hub_repo_cached_file", "hf_hub_download", "model_info"] +) def test_the_sentencepiece_restore_reads_the_stamped_ref(callee): tree = _tree(SAVE) functions = [ diff --git a/unsloth/save.py b/unsloth/save.py index 2113d8e20c..99afc7db76 100644 --- a/unsloth/save.py +++ b/unsloth/save.py @@ -490,9 +490,7 @@ def _has_tokenizer_model(tokenizer, token = None): return False try: - repo_info = HfApi(token = token).model_info( - source, revision = revision, files_metadata = False - ) + repo_info = HfApi(token = token).model_info(source, revision = revision, files_metadata = False) except Exception: return False From 200132780dfc60535444fa945ee27dffda4ce68b Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sun, 2 Aug 2026 18:11:15 +0000 Subject: [PATCH 20/20] Stamp the loaded ref on the vision processor as well FastBaseModel builds its processor without going through load_correct_tokenizer, so the stamp save.py reads was only being applied on the text path and a pinned FastVisionModel load still restored tokenizer.model from the default branch. Stamped at the return rather than at each of the processor branches, so the AutoTokenizer fallback that runs when patch_tokenizer raises cannot lose it either. --- tests/python/test_revision_forwarding.py | 16 ++++++++++++++++ unsloth/models/vision.py | 5 +++++ 2 files changed, 21 insertions(+) diff --git a/tests/python/test_revision_forwarding.py b/tests/python/test_revision_forwarding.py index 3f529b5879..b42604103c 100644 --- a/tests/python/test_revision_forwarding.py +++ b/tests/python/test_revision_forwarding.py @@ -861,3 +861,19 @@ def test_the_sentencepiece_restore_reads_the_stamped_ref(callee): assert calls, f"{callee} not found on the save path" for call in calls: assert _revision_kwarg(call) is not None, f"{callee} at line {call.lineno} drops the ref" + + +def test_the_vision_path_stamps_its_pinned_tokenizer_too(): + """FastBaseModel builds its processor without load_correct_tokenizer, so the stamp the + save path reads has to be applied here as well or a pinned VLM load saves the default + branch's tokenizer.model. At the return, so a patch fallback cannot lose it.""" + function = _function(_tree(VISION), "from_pretrained", "FastBaseModel") + stamps = _calls(function, "_mark_loaded_revision") + assert stamps, "the vision path never stamps its loaded ref" + for call in stamps: + assert any( + getattr(a, "id", None) == "_tokenizer_revision" for a in call.args + ), "stamp the ref the tokenizer was actually read at" + returns = [n for n in ast.walk(function) if isinstance(n, ast.Return)] + assert returns + assert max(c.lineno for c in stamps) < max(r.lineno for r in returns) diff --git a/unsloth/models/vision.py b/unsloth/models/vision.py index d7efa55d84..6ee89f06e9 100644 --- a/unsloth/models/vision.py +++ b/unsloth/models/vision.py @@ -636,6 +636,7 @@ def unsloth_base_fast_generate(self, *args, **kwargs): _hub_repo_or_local_path, _is_offline_related_error, _load_pretrained_tokenizer_fast, + _mark_loaded_revision, _offline_aware_load, ) @@ -1775,6 +1776,10 @@ def _last_resort_tokenizer(lfo): for _ in range(3): gc.collect() clean_gpu_cache() + # Saving restores sentencepiece assets from the repo name alone, which does not + # carry the branch this was read at. Stamped here rather than at each of the + # processor branches above, so a patch fallback cannot lose it. + _mark_loaded_revision(tokenizer, _tokenizer_revision) return model, tokenizer @staticmethod