Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
340 commits
Select commit Hold shift + click to select a range
b4492eb
Update fast_lora.py
danielhanchen Jan 26, 2024
ce08a15
Update fast_lora.py
danielhanchen Jan 26, 2024
485d54f
Update fast_lora.py
danielhanchen Jan 26, 2024
38bff80
Update fast_lora.py
danielhanchen Jan 26, 2024
f776071
Update fast_lora.py
danielhanchen Jan 26, 2024
0a1aa98
Update fast_lora.py
danielhanchen Jan 26, 2024
c74d3dd
Update fast_lora.py
danielhanchen Jan 26, 2024
4ae3ad3
Update fast_lora.py
danielhanchen Jan 26, 2024
f3da0d2
Update fast_lora.py
danielhanchen Jan 26, 2024
83ceb11
Update fast_lora.py
danielhanchen Jan 26, 2024
421ed33
Update fast_lora.py
danielhanchen Jan 26, 2024
d379bb8
Update fast_lora.py
danielhanchen Jan 26, 2024
c57495d
Update fast_lora.py
danielhanchen Jan 26, 2024
ba847f5
Update fast_lora.py
danielhanchen Jan 26, 2024
e3bd0bb
Update save.py
danielhanchen Jan 26, 2024
7fb64e0
Update fast_lora.py
danielhanchen Jan 26, 2024
a27ac61
Update utils.py
danielhanchen Jan 26, 2024
9edc309
Update llama.py
danielhanchen Jan 26, 2024
47babc7
Update fast_lora.py
danielhanchen Jan 26, 2024
393341f
Update swiglu.py
danielhanchen Jan 26, 2024
ecdbb28
Update save.py
danielhanchen Jan 26, 2024
baeea64
Update save.py
danielhanchen Jan 26, 2024
b89599a
Update llama.py
danielhanchen Jan 26, 2024
a3d892a
Update llama.py
danielhanchen Jan 26, 2024
bd2ff90
Update llama.py
danielhanchen Jan 26, 2024
a208ec4
Update llama.py
danielhanchen Jan 26, 2024
704e36a
Revert "Update llama.py"
danielhanchen Jan 26, 2024
a59ec79
Merge branch 'main' into nightly
danielhanchen Jan 26, 2024
fb53337
Update llama.py
danielhanchen Jan 26, 2024
af65cb0
Works?
danielhanchen Jan 26, 2024
8ed03f5
Update pyproject.toml
danielhanchen Jan 27, 2024
f7d11d1
Update fast_lora.py
danielhanchen Jan 27, 2024
e77d7c0
Update fast_lora.py
danielhanchen Jan 27, 2024
d3f3b6f
Update fast_lora.py
danielhanchen Jan 27, 2024
3d3e7f5
Update fast_lora.py
danielhanchen Jan 27, 2024
83b6937
Update fast_lora.py
danielhanchen Jan 27, 2024
85e87d9
Swiglu
danielhanchen Jan 27, 2024
2e4c59d
Update swiglu.py
danielhanchen Jan 27, 2024
8e0e4cc
Update fast_lora.py
danielhanchen Jan 27, 2024
86a1c97
Update fast_lora.py
danielhanchen Jan 27, 2024
6201f76
Update fast_lora.py
danielhanchen Jan 27, 2024
35daafd
Update fast_lora.py
danielhanchen Jan 27, 2024
510c85f
Update swiglu.py
danielhanchen Jan 27, 2024
363ffba
Update fast_lora.py
danielhanchen Jan 27, 2024
c74e1af
Update fast_lora.py
danielhanchen Jan 27, 2024
e094af8
Update fast_lora.py
danielhanchen Jan 27, 2024
6fa0635
Update fast_lora.py
danielhanchen Jan 27, 2024
d01ba45
Update fast_lora.py
danielhanchen Jan 27, 2024
a2f705d
Update fast_lora.py
danielhanchen Jan 27, 2024
a1e5aca
Update fast_lora.py
danielhanchen Jan 27, 2024
2bd77e7
Update fast_lora.py
danielhanchen Jan 27, 2024
9f9739c
attention_mask
danielhanchen Jan 27, 2024
6a027a8
Update llama.py
danielhanchen Jan 27, 2024
4c5ebcc
Update llama.py
danielhanchen Jan 27, 2024
12d57e5
labels
danielhanchen Jan 27, 2024
c836ed7
Update mistral.py
danielhanchen Jan 27, 2024
6c7f0db
Update llama.py
danielhanchen Jan 27, 2024
166f8c8
attention mask
danielhanchen Jan 27, 2024
917ce15
Merge branch 'main' into nightly
danielhanchen Jan 27, 2024
36829b7
Update save.py
danielhanchen Jan 27, 2024
a158003
Merge branch 'nightly' of https://github.com/unslothai/unsloth into n…
danielhanchen Jan 27, 2024
7663a32
Update save.py
danielhanchen Jan 27, 2024
f1b0fd0
Update mistral.py
danielhanchen Jan 28, 2024
ee6f509
attention mask
danielhanchen Jan 28, 2024
6aa46ff
Update llama.py
danielhanchen Jan 28, 2024
2f73cb4
Update llama.py
danielhanchen Jan 28, 2024
2d64e0a
Update mistral.py
danielhanchen Jan 28, 2024
5fe166d
Update llama.py
danielhanchen Jan 28, 2024
ddb48ef
Update llama.py
danielhanchen Jan 28, 2024
d5c852e
Update llama.py
danielhanchen Jan 28, 2024
893aab0
Update dpo.py
danielhanchen Jan 28, 2024
31222ce
Merge branch 'main' into nightly
danielhanchen Jan 28, 2024
788e695
Patch saving
danielhanchen Jan 28, 2024
20e524a
Update save.py
danielhanchen Jan 28, 2024
ac02ba6
Update save.py
danielhanchen Jan 28, 2024
c69c166
patch_saving_functions
danielhanchen Jan 28, 2024
ee0bf6f
Update save.py
danielhanchen Jan 28, 2024
9dec4b3
Update save.py
danielhanchen Jan 28, 2024
fef0589
Update save.py
danielhanchen Jan 28, 2024
74b6977
Update save.py
danielhanchen Jan 28, 2024
b060d7b
Update save.py
danielhanchen Jan 28, 2024
460de24
Update save.py
danielhanchen Jan 28, 2024
9e00cc2
Update save.py
danielhanchen Jan 28, 2024
a3d2b9b
Update save.py
danielhanchen Jan 28, 2024
498dfb8
print
danielhanchen Jan 28, 2024
11ba2c5
Mistral patch
danielhanchen Jan 28, 2024
5bd916b
Update mistral.py
danielhanchen Jan 28, 2024
e10e488
Update save.py
danielhanchen Jan 28, 2024
01e5c30
Merge branch 'main' into nightly
danielhanchen Jan 28, 2024
03ca52d
saving
danielhanchen Jan 28, 2024
25a88ea
Update llama.py
danielhanchen Jan 29, 2024
5cfea20
Update llama.py
danielhanchen Jan 29, 2024
01b8162
Merge branch 'main' into nightly
danielhanchen Jan 29, 2024
58cabcb
Merge branch 'main' into nightly
danielhanchen Jan 29, 2024
4700d51
Fast inference repatch
danielhanchen Jan 29, 2024
bb36420
Update llama.py
danielhanchen Jan 29, 2024
7c87d60
Update utils.py
danielhanchen Jan 29, 2024
a7bfeec
Update utils.py
danielhanchen Jan 29, 2024
6f74c98
Update utils.py
danielhanchen Jan 29, 2024
3ddda6f
Update mistral.py
danielhanchen Jan 29, 2024
71725ae
Update __init__.py
danielhanchen Jan 29, 2024
e0bad0e
Fix inference
danielhanchen Jan 29, 2024
ed5a653
Update mistral.py
danielhanchen Jan 29, 2024
b8f665b
fast lm_head
danielhanchen Jan 30, 2024
270df81
Remove fast path
danielhanchen Jan 30, 2024
4416837
Update rope_embedding.py
danielhanchen Jan 30, 2024
c68a3bc
Update loader.py
danielhanchen Jan 30, 2024
248887b
LlamaAttention_fast_forward_inference
danielhanchen Jan 30, 2024
55fe605
if past_key_value is not None and q_len == 1:
danielhanchen Jan 30, 2024
c0edaa4
revert inference
danielhanchen Jan 30, 2024
d347db0
Update loader.py
danielhanchen Jan 30, 2024
5da0555
past_key_value
danielhanchen Jan 30, 2024
c928c57
Merge branch 'main' into nightly
danielhanchen Jan 31, 2024
5edad55
Update llama.py
danielhanchen Jan 31, 2024
7227de4
Update llama.py
danielhanchen Jan 31, 2024
9f31254
Fix SDPA
danielhanchen Jan 31, 2024
0dc26ed
Update llama.py
danielhanchen Jan 31, 2024
acbdef7
padding
danielhanchen Jan 31, 2024
713a95c
Inference
danielhanchen Jan 31, 2024
329f80a
Update llama.py
danielhanchen Jan 31, 2024
e2f72fe
Revert
danielhanchen Jan 31, 2024
20f1939
Update mistral.py
danielhanchen Jan 31, 2024
648c79e
faster inference
danielhanchen Jan 31, 2024
cf4b58e
inference
danielhanchen Feb 1, 2024
5704450
Update llama.py
danielhanchen Feb 1, 2024
e90b3bf
Update llama.py
danielhanchen Feb 1, 2024
3f0ddf0
Update llama.py
danielhanchen Feb 1, 2024
e2f0dd8
Update llama.py
danielhanchen Feb 1, 2024
a5ee70b
Update llama.py
danielhanchen Feb 1, 2024
a2cb7a1
Update llama.py
danielhanchen Feb 1, 2024
e8ec80a
Update llama.py
danielhanchen Feb 1, 2024
38b5982
Update llama.py
danielhanchen Feb 1, 2024
8920caf
inference
danielhanchen Feb 1, 2024
b83cea7
Update llama.py
danielhanchen Feb 1, 2024
19fb50e
Update utils.py
danielhanchen Feb 1, 2024
1793a16
faster inference
danielhanchen Feb 1, 2024
e791db9
Update llama.py
danielhanchen Feb 1, 2024
24c4c37
revert
danielhanchen Feb 1, 2024
cd39f61
lm_head
danielhanchen Feb 1, 2024
334c5ed
Update llama.py
danielhanchen Feb 1, 2024
e4b5e38
inference
danielhanchen Feb 1, 2024
73f63d6
Update llama.py
danielhanchen Feb 1, 2024
e0ea238
Update llama.py
danielhanchen Feb 1, 2024
5f3c51b
Update llama.py
danielhanchen Feb 1, 2024
f231c4f
Update llama.py
danielhanchen Feb 1, 2024
abc4783
Update llama.py
danielhanchen Feb 1, 2024
da64d34
Update llama.py
danielhanchen Feb 1, 2024
0b661a2
Update llama.py
danielhanchen Feb 1, 2024
7ad1a1f
Update llama.py
danielhanchen Feb 1, 2024
6cc9835
Update llama.py
danielhanchen Feb 2, 2024
40e8848
Update llama.py
danielhanchen Feb 2, 2024
ca99d7c
Update mistral.py
danielhanchen Feb 2, 2024
7c2b042
Update llama.py
danielhanchen Feb 2, 2024
bf44105
faster inference
danielhanchen Feb 2, 2024
168ded9
Update llama.py
danielhanchen Feb 2, 2024
0e1d67d
fast inference
danielhanchen Feb 2, 2024
f299c9c
Update llama.py
danielhanchen Feb 2, 2024
c3c454a
Update llama.py
danielhanchen Feb 2, 2024
1497521
Update mistral.py
danielhanchen Feb 2, 2024
82ead80
Update llama.py
danielhanchen Feb 2, 2024
a5a123d
Update llama.py
danielhanchen Feb 2, 2024
9146fa4
Update llama.py
danielhanchen Feb 2, 2024
e748078
Update llama.py
danielhanchen Feb 2, 2024
82eea75
torch compile
danielhanchen Feb 2, 2024
9df19ae
past_key_values
danielhanchen Feb 2, 2024
1b27421
Update llama.py
danielhanchen Feb 2, 2024
b370392
Update llama.py
danielhanchen Feb 2, 2024
923c6ba
Update llama.py
danielhanchen Feb 2, 2024
b33c92d
Update llama.py
danielhanchen Feb 2, 2024
dc27404
Update llama.py
danielhanchen Feb 2, 2024
03df291
Update llama.py
danielhanchen Feb 2, 2024
c6ad936
Update llama.py
danielhanchen Feb 2, 2024
ea9b4ee
Update llama.py
danielhanchen Feb 2, 2024
9c2bed3
Update llama.py
danielhanchen Feb 2, 2024
705bbba
Update llama.py
danielhanchen Feb 2, 2024
c934c16
Update llama.py
danielhanchen Feb 2, 2024
e911047
Update llama.py
danielhanchen Feb 2, 2024
0be4640
Update llama.py
danielhanchen Feb 2, 2024
a15ffc9
Update llama.py
danielhanchen Feb 2, 2024
4436a10
Update llama.py
danielhanchen Feb 2, 2024
a81d193
Update utils.py
danielhanchen Feb 2, 2024
404e177
Update utils.py
danielhanchen Feb 2, 2024
8e3f029
Update utils.py
danielhanchen Feb 2, 2024
fa3d234
Update utils.py
danielhanchen Feb 2, 2024
a78d6fb
Update llama.py
danielhanchen Feb 2, 2024
ad357de
fast inference + saving config.json
danielhanchen Feb 2, 2024
522f6db
Update llama.py
danielhanchen Feb 3, 2024
dd03abe
Update llama.py
danielhanchen Feb 3, 2024
1c6e1f1
Update llama.py
danielhanchen Feb 3, 2024
8270821
Update llama.py
danielhanchen Feb 3, 2024
9225dd6
Update llama.py
danielhanchen Feb 3, 2024
d76f583
Update mistral.py
danielhanchen Feb 3, 2024
68db1c7
fast inference again
danielhanchen Feb 3, 2024
5534f8a
more temp matrices
danielhanchen Feb 3, 2024
665908e
Update llama.py
danielhanchen Feb 3, 2024
381b991
Update llama.py
danielhanchen Feb 3, 2024
cf0fae9
Update llama.py
danielhanchen Feb 3, 2024
e500b78
Update llama.py
danielhanchen Feb 3, 2024
257cd7d
Update llama.py
danielhanchen Feb 3, 2024
31578f2
fast inference
danielhanchen Feb 3, 2024
aa032fc
Update mistral.py
danielhanchen Feb 3, 2024
fcb8846
Update llama.py
danielhanchen Feb 3, 2024
74d7fc6
SDPA
danielhanchen Feb 3, 2024
711e5c0
attention_mask
danielhanchen Feb 3, 2024
54802ec
New version
danielhanchen Feb 3, 2024
607dfa1
Update llama.py
danielhanchen Feb 3, 2024
8c8685e
Update llama.py
danielhanchen Feb 3, 2024
ac9bc79
Update llama.py
danielhanchen Feb 3, 2024
88a695d
Update llama.py
danielhanchen Feb 3, 2024
4a5d3b1
Update llama.py
danielhanchen Feb 3, 2024
36b400e
Update llama.py
danielhanchen Feb 3, 2024
80fa8e9
Update llama.py
danielhanchen Feb 3, 2024
65270ce
Update llama.py
danielhanchen Feb 3, 2024
9a5ebef
Update llama.py
danielhanchen Feb 3, 2024
d867b9b
Update llama.py
danielhanchen Feb 3, 2024
7166b11
Update llama.py
danielhanchen Feb 3, 2024
63816fc
Update llama.py
danielhanchen Feb 3, 2024
24431b4
Update llama.py
danielhanchen Feb 3, 2024
9cd2517
Update llama.py
danielhanchen Feb 3, 2024
71899d7
Update llama.py
danielhanchen Feb 3, 2024
3bfb0eb
Update llama.py
danielhanchen Feb 3, 2024
00242d5
Update llama.py
danielhanchen Feb 3, 2024
201d90c
Update llama.py
danielhanchen Feb 3, 2024
7ab3426
Update llama.py
danielhanchen Feb 3, 2024
6ab4019
Update llama.py
danielhanchen Feb 3, 2024
990068b
Update utils.py
danielhanchen Feb 4, 2024
63ed23a
Update utils.py
danielhanchen Feb 4, 2024
1750b13
Merge branch 'main' into nightly
danielhanchen Feb 4, 2024
d69cef9
Update save.py
danielhanchen Feb 5, 2024
a50daa1
Update save.py
danielhanchen Feb 5, 2024
03ed3a8
Torch 2.2.0
danielhanchen Feb 5, 2024
33192d6
Update save.py
danielhanchen Feb 5, 2024
e9031ce
mistral swa
danielhanchen Feb 5, 2024
797a87a
Update save.py
danielhanchen Feb 5, 2024
e487abd
Update save.py
danielhanchen Feb 5, 2024
53ca91f
Update save.py
danielhanchen Feb 5, 2024
262289d
Update save.py
danielhanchen Feb 5, 2024
aa0427f
Update save.py
danielhanchen Feb 5, 2024
39a2a7c
Fix SWA inference
danielhanchen Feb 6, 2024
d0b1144
Fix llm_int8_skip_modules
danielhanchen Feb 6, 2024
8e9d9c3
SWA inference
danielhanchen Feb 6, 2024
8b52dc0
Merge branch 'main' into nightly
danielhanchen Feb 6, 2024
213cfee
Update save.py
danielhanchen Feb 6, 2024
bfb3ea7
Update save.py
danielhanchen Feb 6, 2024
9c3849f
Update pyproject.toml
danielhanchen Feb 6, 2024
9e2b00e
__version__
danielhanchen Feb 6, 2024
7da7afc
__version__
danielhanchen Feb 6, 2024
17a6b12
Update save.py
danielhanchen Feb 6, 2024
2b346dc
Update save.py
danielhanchen Feb 6, 2024
ab0a3c9
Update mistral.py
danielhanchen Feb 6, 2024
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
51 changes: 50 additions & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -23,7 +23,7 @@ classifiers = [
]

[tool.setuptools.dynamic]
version = {attr = "unsloth.__version__"}
version = {attr = "unsloth.models._utils.__version__"}

[tool.setuptools]
include-package-data = false
Expand Down Expand Up @@ -62,6 +62,16 @@ cu121onlytorch211 = [
"xformers @ https://download.pytorch.org/whl/cu121/xformers-0.0.23-cp310-cp310-manylinux2014_x86_64.whl ; python_version=='3.10'",
"xformers @ https://download.pytorch.org/whl/cu121/xformers-0.0.23-cp311-cp311-manylinux2014_x86_64.whl ; python_version=='3.11'",
]
cu118onlytorch220 = [
"xformers @ https://download.pytorch.org/whl/cu118/xformers-0.0.24%2Bcu118-cp39-cp39-manylinux2014_x86_64.whl ; python_version=='3.9'",
"xformers @ https://download.pytorch.org/whl/cu118/xformers-0.0.24%2Bcu118-cp310-cp310-manylinux2014_x86_64.whl ; python_version=='3.10'",
"xformers @ https://download.pytorch.org/whl/cu118/xformers-0.0.24%2Bcu118-cp311-cp311-manylinux2014_x86_64.whl ; python_version=='3.11'",
]
cu121onlytorch220 = [
"xformers @ https://download.pytorch.org/whl/cu121/xformers-0.0.24-cp39-cp39-manylinux2014_x86_64.whl ; python_version=='3.9'",
"xformers @ https://download.pytorch.org/whl/cu121/xformers-0.0.24-cp310-cp310-manylinux2014_x86_64.whl ; python_version=='3.10'",
"xformers @ https://download.pytorch.org/whl/cu121/xformers-0.0.24-cp311-cp311-manylinux2014_x86_64.whl ; python_version=='3.11'",
]
cu118 = [
"unsloth[huggingface]",
"bitsandbytes",
Expand All @@ -82,6 +92,16 @@ cu121_torch211 = [
"bitsandbytes",
"unsloth[cu121onlytorch211]",
]
cu118_torch220 = [
"unsloth[huggingface]",
"bitsandbytes",
"unsloth[cu118onlytorch220]",
]
cu121_torch220 = [
"unsloth[huggingface]",
"bitsandbytes",
"unsloth[cu121onlytorch220]",
]
kaggle = [
"unsloth[huggingface]",
]
Expand Down Expand Up @@ -110,6 +130,19 @@ colab_ampere_torch211 = [
"ninja",
"flash-attn",
]
colab_torch220 = [
"unsloth[huggingface]",
"bitsandbytes",
"unsloth[cu121onlytorch220]",
]
colab_ampere_torch220 = [
"unsloth[huggingface]",
"bitsandbytes",
"unsloth[cu121onlytorch220]",
"packaging",
"ninja",
"flash-attn",
]
cu118_ampere = [
"unsloth[huggingface]",
"bitsandbytes",
Expand Down Expand Up @@ -142,6 +175,22 @@ cu121_ampere_torch211 = [
"ninja",
"flash-attn",
]
cu118_ampere_torch220 = [
"unsloth[huggingface]",
"bitsandbytes",
"unsloth[cu118onlytorch220]",
"packaging",
"ninja",
"flash-attn",
]
cu121_ampere_torch220 = [
"unsloth[huggingface]",
"bitsandbytes",
"unsloth[cu121onlytorch220]",
"packaging",
"ninja",
"flash-attn",
]

[project.urls]
homepage = "http://www.unsloth.ai"
Expand Down
1 change: 0 additions & 1 deletion unsloth/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,6 @@
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
__version__ = "2024.1"
import os
import warnings
import importlib
Expand Down
23 changes: 18 additions & 5 deletions unsloth/models/llama.py
Original file line number Diff line number Diff line change
Expand Up @@ -171,15 +171,28 @@ def LlamaAttention_fast_forward_inference(
Kn = self.paged_attention_K[:kv_seq_len].permute(1, 2, 0, 3)
Vn = self.paged_attention_V[:kv_seq_len].permute(1, 2, 0, 3)

# Handle sliding windows
sliding_window = getattr(self.config, "sliding_window", None)
if sliding_window is not None and kv_seq_len > sliding_window:
# From https://github.com/huggingface/transformers/blob/main/src/transformers/models/mistral/modeling_mistral.py#L193
slicing_tokens = 1 - sliding_window
Knn = Kn[:, :, slicing_tokens:, :]#.contiguous()
Vnn = Vn[:, :, slicing_tokens:, :]#.contiguous()
else:
Knn, Vnn = Kn, Vn
pass

# Grouped query attention
if n_groups != 1:
_, _, cached_len, _ = Kn.shape
Knn = Kn[:, :, None, :, :].expand(bsz, n_kv_heads, n_groups, cached_len, head_dim)
Vnn = Vn[:, :, None, :, :].expand(bsz, n_kv_heads, n_groups, cached_len, head_dim)
_, _, cached_len, _ = Knn.shape
Knn = Knn[:, :, None, :, :].expand(bsz, n_kv_heads, n_groups, cached_len, head_dim)
Vnn = Vnn[:, :, None, :, :].expand(bsz, n_kv_heads, n_groups, cached_len, head_dim)
Knn = Knn.reshape(bsz, n_heads, cached_len, head_dim)
Vnn = Vnn.reshape(bsz, n_heads, cached_len, head_dim)
else:
Knn, Vnn = Kn, Vn
pass
# else:
# Knn, Vnn = Knn, Vnn
# pass

# Attention
A = torch.matmul(Qn, Knn.transpose(2, 3), out = self.attention[:,:,:,:kv_seq_len])
Expand Down
2 changes: 1 addition & 1 deletion unsloth/models/loader.py
Original file line number Diff line number Diff line change
Expand Up @@ -128,7 +128,7 @@ def from_pretrained(
"bnb_4bit_use_double_quant" : True,
"llm_int8_enable_fp32_cpu_offload" : False,
"llm_int8_has_fp16_weight" : False,
"llm_int8_skip_modules" : "null",
"llm_int8_skip_modules" : None,
"llm_int8_threshold" : 6.0,
"load_in_4bit" : True,
"load_in_8bit" : False,
Expand Down
18 changes: 13 additions & 5 deletions unsloth/save.py
Original file line number Diff line number Diff line change
Expand Up @@ -72,6 +72,7 @@ def print_quantization_methods():
pass



def _merge_lora(layer, name):

if isinstance(layer, (Bnb_Linear4bit, Peft_Linear4bit, Peft_Linear)):
Expand All @@ -85,9 +86,12 @@ def _merge_lora(layer, name):
W = W.to(torch.float32).t()

if A is not None:
sAB = (A.t().to(torch.float32) @ (s * B.t().to(torch.float32)))
W += sAB
if not torch.isfinite(W).all():
# sAB = (A.t().to(torch.float32) @ (s * B.t().to(torch.float32)))
# W += sAB
W.addmm_(A.t().to(torch.float32), B.t().to(torch.float32), alpha = s)
# if not torch.isfinite(W).all():
maximum_element = torch.max(W.min().abs(), W.max())
if not torch.isfinite(maximum_element).item():
raise ValueError(f"Unsloth: Merge failed.\n{name} has some elements = infinity.")
pass
W = W.t().to(dtype)
Expand Down Expand Up @@ -373,7 +377,7 @@ def unsloth_save_model(
# elif (max_ram - W.nbytes) > 0:
# # Save to CPU memory
# logger.warning_once(f"We will save to RAM and not VRAM now.")
# state_dict[name] = W.to("cpu", non_blocking = True)
# state_dict[name] = W.to("cpu", non_blocking = True, copy = True)
# max_ram = max(max_ram - W.nbytes, 0)
else:
# Save to Disk
Expand Down Expand Up @@ -579,9 +583,11 @@ def save_to_gguf(
f"--outfile {final_location} "\
f"--outtype {first_conversion} --concurrency {n_cpus}"

with subprocess.Popen(command, shell = True, stdout = subprocess.PIPE, bufsize = 1) as sp:
with subprocess.Popen(command, shell = True, stdout = subprocess.PIPE, stderr = subprocess.PIPE, bufsize = 1) as sp:
for line in sp.stdout:
print(line.decode("utf-8"), flush = True, end = "")
if sp.returncode is not None and sp.returncode != 0:
raise subprocess.CalledProcessError(sp.returncode, sp.args)
pass

# Check if quantization succeeded!
Expand Down Expand Up @@ -609,6 +615,8 @@ def save_to_gguf(
with subprocess.Popen(command, shell = True, stderr = subprocess.PIPE, bufsize = 1) as sp:
for line in sp.stderr:
print(line.decode("utf-8"), flush = True, end = "")
if sp.returncode is not None and sp.returncode != 0:
raise subprocess.CalledProcessError(sp.returncode, sp.args)
pass

# Check if quantization succeeded!
Expand Down