tests/studio/install/test_rocm_rdna_routing.py errors out on CPU-only CI,
taking Repo tests (CPU) with it, all 12 cases with
OSError: libhipblas.so.2: cannot open shared object file
AttributeError: module 'torch._C' has no attribute '_cuda_getCurrentRawStream'
The spoof presents torch as a Radeon card, which flips
torch.cuda.is_available() to True and sets torch.version.hip. bitsandbytes
gates its backend on exactly that:
if torch.cuda.is_available():
from .backends.cuda import ops as cuda_ops
so a bitsandbytes imported afterwards walks into the CUDA/ROCm path against a
CPU-only wheel and dies reading torch._C._cuda_getCurrentRawStream. It reaches
the test because unsloth_zoo imports it eagerly, guarded by except ImportError,
which neither OSError nor AttributeError satisfies.
Import it in the spoof instead, while is_available() is still False, so the CPU
path is cached in sys.modules before torch is rewritten. Placed in the shared
apply(), ahead of the first mutation and inside the idempotence guard, so the
ROCm spoof that layers on top gets it too.
Co-authored-by: danielhanchen <unslothai@gmail.com>
220 lines
8 KiB
Python
220 lines
8 KiB
Python
# Auto-generated by .github/workflows/consolidated-tests-ci.yml.
|
|
# Aggressive CUDA spoof for the consolidated CPU-only CI job. Extends
|
|
# tests/conftest.py's harness with deeper patches that unblock more patch_* /
|
|
# unsloth_zoo init paths on a GPU-less runner. Imported by every shim test
|
|
# file before any unsloth / unsloth_zoo / transformers import.
|
|
#
|
|
# Only no-op or value-returning patches; tensor allocators are NOT replaced.
|
|
# The one exception is dropping `pin_memory=True` (meaningless here), which
|
|
# downgrades a CUDA-required call to CPU-OK.
|
|
|
|
from __future__ import annotations
|
|
|
|
import sys
|
|
import types
|
|
from typing import Any
|
|
|
|
|
|
def apply() -> None:
|
|
"""Apply the spoof. Idempotent: calling again has no effect."""
|
|
import torch
|
|
|
|
if getattr(torch.cuda, "_unsloth_consolidated_spoof", False):
|
|
return
|
|
|
|
# Settle bitsandbytes against the real torch first. Its __init__ does
|
|
# `if torch.cuda.is_available(): from .backends.cuda import ops`, and that
|
|
# module reads torch._C._cuda_getCurrentRawStream at import. On a CPU-only
|
|
# wheel that attribute is absent, so a bitsandbytes imported AFTER this
|
|
# spoof raises AttributeError (or OSError hunting libhipblas for the ROCm
|
|
# spoof) rather than ImportError, which slips past the `except ImportError`
|
|
# guards its importers use. Importing it here, while is_available() is
|
|
# still False, caches the CPU path in sys.modules for everything that
|
|
# follows.
|
|
try:
|
|
import bitsandbytes # noqa: F401
|
|
except Exception:
|
|
pass
|
|
|
|
# Device probes (cheap, value-returning)
|
|
torch.cuda.is_available = lambda: True
|
|
torch.cuda.device_count = lambda: 1
|
|
torch.cuda.current_device = lambda: 0
|
|
torch.cuda.is_initialized = lambda: True
|
|
torch.cuda.set_device = lambda *a, **k: None
|
|
torch.cuda.synchronize = lambda *a, **k: None
|
|
torch.cuda.empty_cache = lambda *a, **k: None
|
|
torch.cuda.get_device_name = lambda *a, **k: "NVIDIA A100-SPOOFED"
|
|
torch.cuda.get_device_capability = lambda *a, **k: (8, 0)
|
|
torch.cuda.is_bf16_supported = lambda *a, **k: True
|
|
torch.cuda._is_in_bad_fork = lambda *a, **k: False # type: ignore[attr-defined]
|
|
|
|
class _Props:
|
|
name = "NVIDIA A100-SPOOFED"
|
|
major = 8
|
|
minor = 0
|
|
total_memory = 80 * 1024**3
|
|
multi_processor_count = 108
|
|
is_integrated = False
|
|
is_multi_gpu_board = False
|
|
|
|
torch.cuda.get_device_properties = lambda *a, **k: _Props() # type: ignore[assignment]
|
|
|
|
# cudart() wrapper
|
|
class _CudaRt:
|
|
@staticmethod
|
|
def cudaMemGetInfo(device: int = 0):
|
|
return (0, 80 * 1024**3)
|
|
|
|
@staticmethod
|
|
def cudaGetDeviceCount(*_a, **_k):
|
|
return 0 # unused on the spoof path
|
|
|
|
@staticmethod
|
|
def cudaSetDevice(*_a, **_k):
|
|
return 0
|
|
|
|
torch.cuda.cudart = lambda: _CudaRt() # type: ignore[assignment]
|
|
|
|
# memory module
|
|
try:
|
|
import torch.cuda.memory as _cuda_memory # type: ignore
|
|
|
|
_cuda_memory.mem_get_info = lambda *a, **k: (0, 80 * 1024**3)
|
|
_cuda_memory.memory_stats = lambda *a, **k: {}
|
|
_cuda_memory.memory_allocated = lambda *a, **k: 0
|
|
_cuda_memory.max_memory_allocated = lambda *a, **k: 0
|
|
_cuda_memory.memory_reserved = lambda *a, **k: 0
|
|
_cuda_memory.max_memory_reserved = lambda *a, **k: 0
|
|
_cuda_memory.reset_peak_memory_stats = lambda *a, **k: None
|
|
except Exception:
|
|
pass
|
|
|
|
# nvtx no-op stub
|
|
nvtx_stub = types.ModuleType("torch.cuda.nvtx")
|
|
nvtx_stub.range_push = lambda *a, **k: None # type: ignore[attr-defined]
|
|
nvtx_stub.range_pop = lambda *a, **k: None # type: ignore[attr-defined]
|
|
nvtx_stub.mark = lambda *a, **k: None # type: ignore[attr-defined]
|
|
sys.modules.setdefault("torch.cuda.nvtx", nvtx_stub)
|
|
torch.cuda.nvtx = nvtx_stub # type: ignore[attr-defined]
|
|
|
|
# random API
|
|
# CRITICAL: torch.manual_seed() calls torch.cuda.manual_seed_all(), so
|
|
# routing the cuda seed APIs back through torch.manual_seed would
|
|
# infinite-recurse. No-op them; CUDA seeding is meaningless on CPU.
|
|
torch.cuda.manual_seed = lambda *a, **k: None # type: ignore[assignment]
|
|
torch.cuda.manual_seed_all = lambda *a, **k: None # type: ignore[assignment]
|
|
# rng_state APIs: return a CPU-shaped placeholder; do NOT route through
|
|
# torch.{get,set}_rng_state (those touch the CPU RNG).
|
|
import torch as _t
|
|
|
|
_empty_rng_state = _t.empty(0, dtype = _t.uint8)
|
|
torch.cuda.get_rng_state = lambda *a, **k: _empty_rng_state.clone() # type: ignore[assignment]
|
|
torch.cuda.set_rng_state = lambda *a, **k: None # type: ignore[assignment]
|
|
torch.cuda.get_rng_state_all = lambda *a, **k: [_empty_rng_state.clone()] # type: ignore[attr-defined]
|
|
torch.cuda.set_rng_state_all = lambda *a, **k: None # type: ignore[attr-defined]
|
|
torch.cuda.initial_seed = lambda *a, **k: 0 # type: ignore[assignment]
|
|
torch.cuda.seed = lambda *a, **k: None # type: ignore[assignment]
|
|
torch.cuda.seed_all = lambda *a, **k: None # type: ignore[assignment]
|
|
|
|
# Stream / Event no-op classes
|
|
class _NoopStream:
|
|
def __init__(self, *a, **k): ...
|
|
def __enter__(self):
|
|
return self
|
|
|
|
def __exit__(self, *a):
|
|
return False
|
|
|
|
def synchronize(self, *a, **k): ...
|
|
def wait_stream(self, *a, **k): ...
|
|
def query(self):
|
|
return True
|
|
|
|
class _NoopEvent:
|
|
def __init__(self, *a, **k): ...
|
|
def record(self, *a, **k): ...
|
|
def wait(self, *a, **k): ...
|
|
def query(self):
|
|
return True
|
|
|
|
def synchronize(self, *a, **k): ...
|
|
def elapsed_time(self, *a, **k):
|
|
return 0.0
|
|
|
|
torch.cuda.Stream = _NoopStream # type: ignore[assignment]
|
|
torch.cuda.Event = _NoopEvent # type: ignore[assignment]
|
|
torch.cuda.stream = lambda s: s if s is not None else _NoopStream() # type: ignore[assignment]
|
|
torch.cuda.current_stream = lambda *a, **k: _NoopStream() # type: ignore[assignment]
|
|
torch.cuda.default_stream = lambda *a, **k: _NoopStream() # type: ignore[assignment]
|
|
|
|
# pin_memory drop: pin_memory=True raises on a CPU-only build; strip the kwarg.
|
|
for _name in (
|
|
"empty",
|
|
"zeros",
|
|
"ones",
|
|
"empty_like",
|
|
"zeros_like",
|
|
"ones_like",
|
|
"rand",
|
|
"randn",
|
|
"randint",
|
|
):
|
|
_orig = getattr(torch, _name, None)
|
|
if _orig is None:
|
|
continue
|
|
|
|
def _wrap(
|
|
*args: Any,
|
|
_orig = _orig,
|
|
**kwargs: Any,
|
|
):
|
|
kwargs.pop("pin_memory", None)
|
|
return _orig(*args, **kwargs)
|
|
|
|
setattr(torch, _name, _wrap)
|
|
|
|
# Tensor.pin_memory() instance method: also a no-op (return self).
|
|
if hasattr(torch.Tensor, "pin_memory"):
|
|
torch.Tensor.pin_memory = lambda self, *a, **k: self # type: ignore[assignment]
|
|
if hasattr(torch.Tensor, "is_pinned"):
|
|
torch.Tensor.is_pinned = lambda self, *a, **k: False # type: ignore[assignment]
|
|
|
|
# amp.GradScaler: use the real one if importable (newer torch handles CPU), else stub.
|
|
try:
|
|
import torch.cuda.amp # type: ignore
|
|
except Exception:
|
|
cuda_amp = types.ModuleType("torch.cuda.amp")
|
|
|
|
class _StubScaler:
|
|
def __init__(self, *a, **k): ...
|
|
def scale(self, x):
|
|
return x
|
|
|
|
def step(self, opt):
|
|
opt.step()
|
|
|
|
def update(self, *a, **k): ...
|
|
def unscale_(self, *a, **k): ...
|
|
def get_scale(self):
|
|
return 1.0
|
|
|
|
def is_enabled(self):
|
|
return False
|
|
|
|
def state_dict(self):
|
|
return {}
|
|
|
|
def load_state_dict(self, *a, **k): ...
|
|
|
|
cuda_amp.GradScaler = _StubScaler # type: ignore[attr-defined]
|
|
sys.modules.setdefault("torch.cuda.amp", cuda_amp)
|
|
torch.cuda.amp = cuda_amp # type: ignore[attr-defined]
|
|
|
|
# Sentinel
|
|
torch.cuda._unsloth_consolidated_spoof = True # type: ignore[attr-defined]
|
|
|
|
|
|
if __name__ == "__main__":
|
|
apply()
|
|
print("CUDA spoof applied.")
|