unsloth/unsloth/device_type.py
Daniel Han 77e0929735
revert: stop touching DEVICE_TYPE == "cuda" branches for CPU CI (#5473)
#5429 (cb15a7a5) tightened three production-path branches to
DEVICE_TYPE == "cuda" and torch.cuda.is_available() and added a
new else: SUPPORTS_BFLOAT16 = False arm to let `import unsloth.trainer`
survive on a CPU-only CI host. We already ship the package on Intel
XPU / AMD HIP / NVIDIA CUDA and don't want any extra branching in
those hot paths.

Move the entire CPU-CI handling to one place -- the top of
unsloth/device_type.get_device_type() -- so the UNSLOTH_ALLOW_CPU=1
sentinel short-circuits detection and returns "cuda" before any
torch probe runs. Every downstream DEVICE_TYPE == "cuda" branch
then behaves identically to a real CUDA host, with no additional
checks. The two existing duplicate UNSLOTH_ALLOW_CPU returns
later in the function are dropped (the new top-of-function check
covers both).

Revert the three call-site changes:
- unsloth/_gpu_init.py:212  -> back to `if DEVICE_TYPE == "cuda":`
- unsloth/_gpu_init.py:247  -> back to `if DEVICE_TYPE == "cuda":`
- unsloth/models/_utils.py:1207 -> back to `if DEVICE_TYPE == "cuda":`
- unsloth/_gpu_init.py: drop the new `else: SUPPORTS_BFLOAT16 = False`
  branch (dead under the top-of-function short-circuit).

Keep the two env-var gates that are needed for zoo's drift detectors
to inspect pristine TRL source (no behavioural change on production
hosts that never set UNSLOTH_ALLOW_CPU):
- unsloth/_gpu_init.py: `if env != "1": _patch_trl_trainer()`
- unsloth/models/rl.py:PatchFastRL: `if env == "1": return`

Verified:
- CUDA_VISIBLE_DEVICES=5 python -c "import unsloth.trainer" produces
  UnslothSFTTrainer.__init__ (TRL still patched on real hosts).
- UNSLOTH_ALLOW_CPU=1 + aggressive cuda spoof import succeeds and
  trl.SFTTrainer.__init__.__qualname__ stays SFTTrainer.__init__.
- pytest tests/_zoo_compiler_cache_shim.py -> 5 passed, 1 skipped.
2026-05-15 19:41:09 -07:00

160 lines
5.5 KiB
Python

# Copyright 2023-present Daniel Han-Chen & the Unsloth team. All rights reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
__all__ = [
"is_hip",
"get_device_type",
"DEVICE_TYPE",
"DEVICE_TYPE_TORCH",
"DEVICE_COUNT",
"ALLOW_PREQUANTIZED_MODELS",
"ALLOW_BITSANDBYTES",
"is_mlx_available",
]
import functools
import inspect
import os
from unsloth_zoo.utils import Version
def is_mlx_available():
try:
from unsloth_zoo.mlx import is_mlx_available as _is_mlx_available
except ImportError:
return False
return _is_mlx_available()
_IS_MLX = is_mlx_available()
if not _IS_MLX:
import torch
@functools.cache
def is_hip():
if _IS_MLX:
return False
return bool(getattr(getattr(torch, "version", None), "hip", None))
@functools.cache
def get_device_type():
# Test-only CPU fallback. Short-circuits the detection chain so the
# rest of the function -- and every DEVICE_TYPE == "cuda" branch in
# the codebase -- behaves identically to a real CUDA host. The env
# var is read exactly once per process because get_device_type is
# @functools.cache'd, so production hosts pay no runtime cost.
if os.environ.get("UNSLOTH_ALLOW_CPU", "0") == "1":
return "cuda"
if _IS_MLX:
return "mlx"
if hasattr(torch, "cuda") and torch.cuda.is_available():
if is_hip():
return "hip"
return "cuda"
elif hasattr(torch, "xpu") and torch.xpu.is_available():
return "xpu"
# Check torch.accelerator
if hasattr(torch, "accelerator"):
if not torch.accelerator.is_available():
raise NotImplementedError(
"Unsloth cannot find any torch accelerator? You need a GPU."
)
accelerator = str(torch.accelerator.current_accelerator())
if accelerator in ("cuda", "xpu", "hip"):
raise RuntimeError(
f"Unsloth: Weirdly `torch.cuda.is_available()`, `torch.xpu.is_available()` and `is_hip` all failed.\n"
f"But `torch.accelerator.current_accelerator()` works with it being = `{accelerator}`\n"
f"Please reinstall torch - it's most likely broken :("
)
raise NotImplementedError(
"Unsloth currently only works on NVIDIA, AMD and Intel GPUs."
)
DEVICE_TYPE: str = get_device_type()
# HIP fails for autocast and other torch functions. Use CUDA instead
DEVICE_TYPE_TORCH = DEVICE_TYPE
if DEVICE_TYPE_TORCH == "hip":
DEVICE_TYPE_TORCH = "cuda"
elif DEVICE_TYPE_TORCH == "mlx":
DEVICE_TYPE_TORCH = "mps"
@functools.cache
def get_device_count():
if DEVICE_TYPE in ("cuda", "hip"):
return torch.cuda.device_count()
elif DEVICE_TYPE == "xpu":
return torch.xpu.device_count()
else:
return 1
DEVICE_COUNT: int = get_device_count()
# 4-bit quantization requires a block size of 64
# | Device Type | Warp Size | Block Size |
# |-----------------|-----------|------------|
# | CUDA | 32 | 32 |
# | Radeon (Navi) | 32 | 32 |
# | Instinct (MI) | 64 | 32 |
#
# Since bitsandbytes 0.49.0, pre-quantized models with 64 blockwise now works
# on Radeon GPUs, but not Instinct MI300x for eg
# See https://github.com/bitsandbytes-foundation/bitsandbytes/pull/1748
#
# Since bitsandbytes 0.49.2, blocksize=64 4-bit quantization is supported on
# CDNA (MI Instinct / gfx9xx) GPUs as well
# See https://github.com/bitsandbytes-foundation/bitsandbytes/pull/1856
ALLOW_PREQUANTIZED_MODELS: bool = True
# HSA_STATUS_ERROR_EXCEPTION checks - sometimes AMD fails for BnB
ALLOW_BITSANDBYTES: bool = True
if DEVICE_TYPE == "hip":
try:
import bitsandbytes
except:
print(
"Unsloth: `bitsandbytes` is not installed - 4bit QLoRA unallowed, but 16bit and full finetuning works."
)
ALLOW_PREQUANTIZED_MODELS = False
ALLOW_BITSANDBYTES = False
if ALLOW_BITSANDBYTES:
ALLOW_BITSANDBYTES = Version(bitsandbytes.__version__) > Version("0.48.2.dev0")
if Version(bitsandbytes.__version__) >= Version("0.49.2"):
pass
elif Version(bitsandbytes.__version__) >= Version("0.49.0"):
try:
# Pre-quantized bitsandbytes models use blocksize 64, so we need to check the GPU
from bitsandbytes.cextension import ROCM_WARP_SIZE_64
ALLOW_PREQUANTIZED_MODELS = not ROCM_WARP_SIZE_64
except Exception as e:
print(
"Unsloth: Checking `from bitsandbytes.cextension import ROCM_WARP_SIZE_64` had error = \n"
f"{str(e)}\n"
"4bit QLoRA disabled for now, but 16bit and full finetuning works."
)
ALLOW_PREQUANTIZED_MODELS = False
ALLOW_BITSANDBYTES = False
elif ALLOW_BITSANDBYTES:
from bitsandbytes.nn.modules import Params4bit
if "blocksize = 64 if not HIP_ENVIRONMENT else 128" in inspect.getsource(
Params4bit
):
ALLOW_PREQUANTIZED_MODELS = False