unsloth/tests/_zoo_aggressive_cuda_spoof.py
Daniel Han 0c1c9f71db
Import bitsandbytes before the hardware spoof rewrites torch (#7471)
tests/studio/install/test_rocm_rdna_routing.py errors out on CPU-only CI,
taking Repo tests (CPU) with it, all 12 cases with

  OSError: libhipblas.so.2: cannot open shared object file
  AttributeError: module 'torch._C' has no attribute '_cuda_getCurrentRawStream'

The spoof presents torch as a Radeon card, which flips
torch.cuda.is_available() to True and sets torch.version.hip. bitsandbytes
gates its backend on exactly that:

  if torch.cuda.is_available():
      from .backends.cuda import ops as cuda_ops

so a bitsandbytes imported afterwards walks into the CUDA/ROCm path against a
CPU-only wheel and dies reading torch._C._cuda_getCurrentRawStream. It reaches
the test because unsloth_zoo imports it eagerly, guarded by except ImportError,
which neither OSError nor AttributeError satisfies.

Import it in the spoof instead, while is_available() is still False, so the CPU
path is cached in sys.modules before torch is rewritten. Placed in the shared
apply(), ahead of the first mutation and inside the idempotence guard, so the
ROCm spoof that layers on top gets it too.

Co-authored-by: danielhanchen <unslothai@gmail.com>
2026-07-26 05:46:12 -07:00

220 lines
8 KiB
Python

# Auto-generated by .github/workflows/consolidated-tests-ci.yml.
# Aggressive CUDA spoof for the consolidated CPU-only CI job. Extends
# tests/conftest.py's harness with deeper patches that unblock more patch_* /
# unsloth_zoo init paths on a GPU-less runner. Imported by every shim test
# file before any unsloth / unsloth_zoo / transformers import.
#
# Only no-op or value-returning patches; tensor allocators are NOT replaced.
# The one exception is dropping `pin_memory=True` (meaningless here), which
# downgrades a CUDA-required call to CPU-OK.
from __future__ import annotations
import sys
import types
from typing import Any
def apply() -> None:
"""Apply the spoof. Idempotent: calling again has no effect."""
import torch
if getattr(torch.cuda, "_unsloth_consolidated_spoof", False):
return
# Settle bitsandbytes against the real torch first. Its __init__ does
# `if torch.cuda.is_available(): from .backends.cuda import ops`, and that
# module reads torch._C._cuda_getCurrentRawStream at import. On a CPU-only
# wheel that attribute is absent, so a bitsandbytes imported AFTER this
# spoof raises AttributeError (or OSError hunting libhipblas for the ROCm
# spoof) rather than ImportError, which slips past the `except ImportError`
# guards its importers use. Importing it here, while is_available() is
# still False, caches the CPU path in sys.modules for everything that
# follows.
try:
import bitsandbytes # noqa: F401
except Exception:
pass
# Device probes (cheap, value-returning)
torch.cuda.is_available = lambda: True
torch.cuda.device_count = lambda: 1
torch.cuda.current_device = lambda: 0
torch.cuda.is_initialized = lambda: True
torch.cuda.set_device = lambda *a, **k: None
torch.cuda.synchronize = lambda *a, **k: None
torch.cuda.empty_cache = lambda *a, **k: None
torch.cuda.get_device_name = lambda *a, **k: "NVIDIA A100-SPOOFED"
torch.cuda.get_device_capability = lambda *a, **k: (8, 0)
torch.cuda.is_bf16_supported = lambda *a, **k: True
torch.cuda._is_in_bad_fork = lambda *a, **k: False # type: ignore[attr-defined]
class _Props:
name = "NVIDIA A100-SPOOFED"
major = 8
minor = 0
total_memory = 80 * 1024**3
multi_processor_count = 108
is_integrated = False
is_multi_gpu_board = False
torch.cuda.get_device_properties = lambda *a, **k: _Props() # type: ignore[assignment]
# cudart() wrapper
class _CudaRt:
@staticmethod
def cudaMemGetInfo(device: int = 0):
return (0, 80 * 1024**3)
@staticmethod
def cudaGetDeviceCount(*_a, **_k):
return 0 # unused on the spoof path
@staticmethod
def cudaSetDevice(*_a, **_k):
return 0
torch.cuda.cudart = lambda: _CudaRt() # type: ignore[assignment]
# memory module
try:
import torch.cuda.memory as _cuda_memory # type: ignore
_cuda_memory.mem_get_info = lambda *a, **k: (0, 80 * 1024**3)
_cuda_memory.memory_stats = lambda *a, **k: {}
_cuda_memory.memory_allocated = lambda *a, **k: 0
_cuda_memory.max_memory_allocated = lambda *a, **k: 0
_cuda_memory.memory_reserved = lambda *a, **k: 0
_cuda_memory.max_memory_reserved = lambda *a, **k: 0
_cuda_memory.reset_peak_memory_stats = lambda *a, **k: None
except Exception:
pass
# nvtx no-op stub
nvtx_stub = types.ModuleType("torch.cuda.nvtx")
nvtx_stub.range_push = lambda *a, **k: None # type: ignore[attr-defined]
nvtx_stub.range_pop = lambda *a, **k: None # type: ignore[attr-defined]
nvtx_stub.mark = lambda *a, **k: None # type: ignore[attr-defined]
sys.modules.setdefault("torch.cuda.nvtx", nvtx_stub)
torch.cuda.nvtx = nvtx_stub # type: ignore[attr-defined]
# random API
# CRITICAL: torch.manual_seed() calls torch.cuda.manual_seed_all(), so
# routing the cuda seed APIs back through torch.manual_seed would
# infinite-recurse. No-op them; CUDA seeding is meaningless on CPU.
torch.cuda.manual_seed = lambda *a, **k: None # type: ignore[assignment]
torch.cuda.manual_seed_all = lambda *a, **k: None # type: ignore[assignment]
# rng_state APIs: return a CPU-shaped placeholder; do NOT route through
# torch.{get,set}_rng_state (those touch the CPU RNG).
import torch as _t
_empty_rng_state = _t.empty(0, dtype = _t.uint8)
torch.cuda.get_rng_state = lambda *a, **k: _empty_rng_state.clone() # type: ignore[assignment]
torch.cuda.set_rng_state = lambda *a, **k: None # type: ignore[assignment]
torch.cuda.get_rng_state_all = lambda *a, **k: [_empty_rng_state.clone()] # type: ignore[attr-defined]
torch.cuda.set_rng_state_all = lambda *a, **k: None # type: ignore[attr-defined]
torch.cuda.initial_seed = lambda *a, **k: 0 # type: ignore[assignment]
torch.cuda.seed = lambda *a, **k: None # type: ignore[assignment]
torch.cuda.seed_all = lambda *a, **k: None # type: ignore[assignment]
# Stream / Event no-op classes
class _NoopStream:
def __init__(self, *a, **k): ...
def __enter__(self):
return self
def __exit__(self, *a):
return False
def synchronize(self, *a, **k): ...
def wait_stream(self, *a, **k): ...
def query(self):
return True
class _NoopEvent:
def __init__(self, *a, **k): ...
def record(self, *a, **k): ...
def wait(self, *a, **k): ...
def query(self):
return True
def synchronize(self, *a, **k): ...
def elapsed_time(self, *a, **k):
return 0.0
torch.cuda.Stream = _NoopStream # type: ignore[assignment]
torch.cuda.Event = _NoopEvent # type: ignore[assignment]
torch.cuda.stream = lambda s: s if s is not None else _NoopStream() # type: ignore[assignment]
torch.cuda.current_stream = lambda *a, **k: _NoopStream() # type: ignore[assignment]
torch.cuda.default_stream = lambda *a, **k: _NoopStream() # type: ignore[assignment]
# pin_memory drop: pin_memory=True raises on a CPU-only build; strip the kwarg.
for _name in (
"empty",
"zeros",
"ones",
"empty_like",
"zeros_like",
"ones_like",
"rand",
"randn",
"randint",
):
_orig = getattr(torch, _name, None)
if _orig is None:
continue
def _wrap(
*args: Any,
_orig = _orig,
**kwargs: Any,
):
kwargs.pop("pin_memory", None)
return _orig(*args, **kwargs)
setattr(torch, _name, _wrap)
# Tensor.pin_memory() instance method: also a no-op (return self).
if hasattr(torch.Tensor, "pin_memory"):
torch.Tensor.pin_memory = lambda self, *a, **k: self # type: ignore[assignment]
if hasattr(torch.Tensor, "is_pinned"):
torch.Tensor.is_pinned = lambda self, *a, **k: False # type: ignore[assignment]
# amp.GradScaler: use the real one if importable (newer torch handles CPU), else stub.
try:
import torch.cuda.amp # type: ignore
except Exception:
cuda_amp = types.ModuleType("torch.cuda.amp")
class _StubScaler:
def __init__(self, *a, **k): ...
def scale(self, x):
return x
def step(self, opt):
opt.step()
def update(self, *a, **k): ...
def unscale_(self, *a, **k): ...
def get_scale(self):
return 1.0
def is_enabled(self):
return False
def state_dict(self):
return {}
def load_state_dict(self, *a, **k): ...
cuda_amp.GradScaler = _StubScaler # type: ignore[attr-defined]
sys.modules.setdefault("torch.cuda.amp", cuda_amp)
torch.cuda.amp = cuda_amp # type: ignore[attr-defined]
# Sentinel
torch.cuda._unsloth_consolidated_spoof = True # type: ignore[attr-defined]
if __name__ == "__main__":
apply()
print("CUDA spoof applied.")