# Auto-generated by .github/workflows/consolidated-tests-ci.yml. # Aggressive CUDA spoof for the consolidated CPU-only CI job. Extends # tests/conftest.py's import-time harness with deeper patches that unblock # more patch_* and unsloth_zoo init paths on a GPU-less runner. Imported by # every shim test file before any unsloth / unsloth_zoo / transformers import. # # Only no-op or value-returning patches; tensor allocators are NOT replaced. # The one exception is dropping `pin_memory=True` (a CUDA-host fast-copy hint # that is meaningless here), which downgrades a CUDA-required call to CPU-OK. from __future__ import annotations import sys import types from typing import Any def apply() -> None: """Apply the spoof. Idempotent: calling again has no effect.""" import torch if getattr(torch.cuda, "_unsloth_consolidated_spoof", False): return # Device probes (cheap, value-returning) torch.cuda.is_available = lambda: True torch.cuda.device_count = lambda: 1 torch.cuda.current_device = lambda: 0 torch.cuda.is_initialized = lambda: True torch.cuda.set_device = lambda *a, **k: None torch.cuda.synchronize = lambda *a, **k: None torch.cuda.empty_cache = lambda *a, **k: None torch.cuda.get_device_name = lambda *a, **k: "NVIDIA A100-SPOOFED" torch.cuda.get_device_capability = lambda *a, **k: (8, 0) torch.cuda.is_bf16_supported = lambda *a, **k: True torch.cuda._is_in_bad_fork = lambda *a, **k: False # type: ignore[attr-defined] class _Props: name = "NVIDIA A100-SPOOFED" major = 8 minor = 0 total_memory = 80 * 1024**3 multi_processor_count = 108 is_integrated = False is_multi_gpu_board = False torch.cuda.get_device_properties = lambda *a, **k: _Props() # type: ignore[assignment] # cudart() wrapper class _CudaRt: @staticmethod def cudaMemGetInfo(device: int = 0): return (0, 80 * 1024**3) @staticmethod def cudaGetDeviceCount(*_a, **_k): return 0 # unused on the spoof path @staticmethod def cudaSetDevice(*_a, **_k): return 0 torch.cuda.cudart = lambda: _CudaRt() # type: ignore[assignment] # memory module try: import torch.cuda.memory as _cuda_memory # type: ignore _cuda_memory.mem_get_info = lambda *a, **k: (0, 80 * 1024**3) _cuda_memory.memory_stats = lambda *a, **k: {} _cuda_memory.memory_allocated = lambda *a, **k: 0 _cuda_memory.max_memory_allocated = lambda *a, **k: 0 _cuda_memory.memory_reserved = lambda *a, **k: 0 _cuda_memory.max_memory_reserved = lambda *a, **k: 0 _cuda_memory.reset_peak_memory_stats = lambda *a, **k: None except Exception: pass # nvtx no-op stub nvtx_stub = types.ModuleType("torch.cuda.nvtx") nvtx_stub.range_push = lambda *a, **k: None # type: ignore[attr-defined] nvtx_stub.range_pop = lambda *a, **k: None # type: ignore[attr-defined] nvtx_stub.mark = lambda *a, **k: None # type: ignore[attr-defined] sys.modules.setdefault("torch.cuda.nvtx", nvtx_stub) torch.cuda.nvtx = nvtx_stub # type: ignore[attr-defined] # random API # CRITICAL: torch.manual_seed() calls torch.cuda.manual_seed_all(), so # routing the cuda seed APIs back through torch.manual_seed would # infinite-recurse (RecursionError in CI). No-op them; CUDA-side seeding # is meaningless on a GPU-less runner. torch.cuda.manual_seed = lambda *a, **k: None # type: ignore[assignment] torch.cuda.manual_seed_all = lambda *a, **k: None # type: ignore[assignment] # rng_state APIs: return a CPU-shaped placeholder, accept anything for set; # do NOT route through torch.{get,set}_rng_state (those touch the CPU RNG). import torch as _t _empty_rng_state = _t.empty(0, dtype = _t.uint8) torch.cuda.get_rng_state = lambda *a, **k: _empty_rng_state.clone() # type: ignore[assignment] torch.cuda.set_rng_state = lambda *a, **k: None # type: ignore[assignment] torch.cuda.get_rng_state_all = lambda *a, **k: [_empty_rng_state.clone()] # type: ignore[attr-defined] torch.cuda.set_rng_state_all = lambda *a, **k: None # type: ignore[attr-defined] torch.cuda.initial_seed = lambda *a, **k: 0 # type: ignore[assignment] torch.cuda.seed = lambda *a, **k: None # type: ignore[assignment] torch.cuda.seed_all = lambda *a, **k: None # type: ignore[assignment] # Stream / Event no-op classes class _NoopStream: def __init__(self, *a, **k): ... def __enter__(self): return self def __exit__(self, *a): return False def synchronize(self, *a, **k): ... def wait_stream(self, *a, **k): ... def query(self): return True class _NoopEvent: def __init__(self, *a, **k): ... def record(self, *a, **k): ... def wait(self, *a, **k): ... def query(self): return True def synchronize(self, *a, **k): ... def elapsed_time(self, *a, **k): return 0.0 torch.cuda.Stream = _NoopStream # type: ignore[assignment] torch.cuda.Event = _NoopEvent # type: ignore[assignment] torch.cuda.stream = lambda s: s if s is not None else _NoopStream() # type: ignore[assignment] torch.cuda.current_stream = lambda *a, **k: _NoopStream() # type: ignore[assignment] torch.cuda.default_stream = lambda *a, **k: _NoopStream() # type: ignore[assignment] # pin_memory drop: torch.empty(..., pin_memory=True) and friends raise on # a CPU-only build; strip the kwarg since pin_memory has no meaning here. for _name in ( "empty", "zeros", "ones", "empty_like", "zeros_like", "ones_like", "rand", "randn", "randint", ): _orig = getattr(torch, _name, None) if _orig is None: continue def _wrap( *args: Any, _orig = _orig, **kwargs: Any, ): kwargs.pop("pin_memory", None) return _orig(*args, **kwargs) setattr(torch, _name, _wrap) # Tensor.pin_memory() instance method: also a no-op (return self). if hasattr(torch.Tensor, "pin_memory"): torch.Tensor.pin_memory = lambda self, *a, **k: self # type: ignore[assignment] if hasattr(torch.Tensor, "is_pinned"): torch.Tensor.is_pinned = lambda self, *a, **k: False # type: ignore[assignment] # amp.GradScaler: use the real one if importable (newer torch handles CPU), # else stub. try: import torch.cuda.amp # type: ignore except Exception: cuda_amp = types.ModuleType("torch.cuda.amp") class _StubScaler: def __init__(self, *a, **k): ... def scale(self, x): return x def step(self, opt): opt.step() def update(self, *a, **k): ... def unscale_(self, *a, **k): ... def get_scale(self): return 1.0 def is_enabled(self): return False def state_dict(self): return {} def load_state_dict(self, *a, **k): ... cuda_amp.GradScaler = _StubScaler # type: ignore[attr-defined] sys.modules.setdefault("torch.cuda.amp", cuda_amp) torch.cuda.amp = cuda_amp # type: ignore[attr-defined] # Sentinel torch.cuda._unsloth_consolidated_spoof = True # type: ignore[attr-defined] if __name__ == "__main__": apply() print("CUDA spoof applied.")