mirror of
https://github.com/furyhawk/home_stack.git
synced 2026-07-21 10:16:47 +00:00
- Introduced a new Jupyter notebook `unsloth_nb.ipynb` for utilizing the Unsloth library. - Implemented code to load and patch FastLanguageModel and FastQwen2Model for improved training speed. - Configured models to support 4-bit quantization for reduced memory usage. - Included detailed logging for model loading and system specifications.
5.6 KiB
5.6 KiB
In [2]:
from unsloth import FastLanguageModel
import torch🦥 Unsloth: Will patch your computer to enable 2x faster free finetuning.
/home/user/projects/ai_stack/backend/.venv/lib/python3.10/site-packages/tqdm/auto.py:21: TqdmWarning: IProgress not found. Please update jupyter and ipywidgets. See https://ipywidgets.readthedocs.io/en/stable/user_install.html from .autonotebook import tqdm as notebook_tqdm
🦥 Unsloth Zoo will now patch everything to make training faster!
In [ ]:
model, tokenizer = FastLanguageModel.from_pretrained(
model_name = "unsloth/llama-3.1-8b-bnb-4bit", # Well-tested model
max_seq_length = 2048, # Context length - can be longer, but uses more memory
load_in_4bit = True, # 4bit uses much less memory
load_in_8bit = False, # A bit more accurate, uses 2x memory
full_finetuning = False, # We have full finetuning now!
# token = "hf_...", # use one if using gated models
)==((====))== Unsloth 2025.5.9: Fast Llama patching. Transformers: 4.52.4. \\ /| NVIDIA RTX 3500 Ada Generation Laptop GPU. Num GPUs = 1. Max memory: 11.607 GB. Platform: Linux. O^O/ \_/ \ Torch: 2.7.0+cu126. CUDA: 8.9. CUDA Toolkit: 12.6. Triton: 3.3.0 \ / Bfloat16 = TRUE. FA [Xformers = 0.0.30. FA2 = False] "-____-" Free license: http://github.com/unslothai/unsloth Unsloth: Fast downloading is enabled - ignore downloading bars which are red colored!
In [1]:
from unsloth import FastQwen2Model
import torch
max_seq_length = 2048 # Choose any! We auto support RoPE Scaling internally!
dtype = None # None for auto detection. Float16 for Tesla T4, V100, Bfloat16 for Ampere+
load_in_4bit = True # Use 4bit quantization to reduce memory usage. Can be False.
# 4bit pre quantized models we support for 4x faster downloading + no OOMs.
fourbit_models = [
"unsloth/Meta-Llama-3.1-8B-bnb-4bit", # Llama-3.1 2x faster
"unsloth/Meta-Llama-3.1-70B-bnb-4bit",
"unsloth/Mistral-Small-Instruct-2409", # Mistral 22b 2x faster!
"unsloth/mistral-7b-instruct-v0.3-bnb-4bit",
"unsloth/Phi-3.5-mini-instruct", # Phi-3.5 2x faster!
"unsloth/Phi-3-medium-4k-instruct",
"unsloth/gemma-2-27b-bnb-4bit", # Gemma 2x faster!
"unsloth/Llama-3.2-1B-bnb-4bit", # NEW! Llama 3.2 models
"unsloth/Llama-3.2-1B-Instruct-bnb-4bit",
"unsloth/Llama-3.2-3B-Instruct-bnb-4bit",
] # More models at https://huggingface.co/unsloth
qwen_models = [
"unsloth/Qwen2.5-Coder-32B-Instruct", # Qwen 2.5 Coder 2x faster
"unsloth/Qwen2.5-Coder-7B",
"unsloth/Qwen2.5-14B-Instruct", # 14B fits in a 16GB card
"unsloth/Qwen2.5-7B",
"unsloth/Qwen2.5-72B-Instruct", # 72B fits in a 48GB card
] # More models at https://huggingface.co/unsloth
model, tokenizer = FastQwen2Model.from_pretrained(
model_name="unsloth/Qwen2.5-Coder-1.5B-Instruct",
max_seq_length=None,
dtype=None,
load_in_4bit=False,
fix_tokenizer=False
# token = "hf_...", # use one if using gated models like meta-llama/Llama-2-7b-hf
)🦥 Unsloth: Will patch your computer to enable 2x faster free finetuning. 🦥 Unsloth Zoo will now patch everything to make training faster! ==((====))== Unsloth 2025.5.9: Fast Qwen2 patching. Transformers: 4.52.4. \\ /| NVIDIA RTX 3500 Ada Generation Laptop GPU. Num GPUs = 1. Max memory: 11.607 GB. Platform: Linux. O^O/ \_/ \ Torch: 2.7.0+cu126. CUDA: 8.9. CUDA Toolkit: 12.6. Triton: 3.3.0 \ / Bfloat16 = TRUE. FA [Xformers = 0.0.30. FA2 = False] "-____-" Free license: http://github.com/unslothai/unsloth Unsloth: Fast downloading is enabled - ignore downloading bars which are red colored!
In [ ]: