Add Intel XPU support to Unsloth Studio (#4724)
--------- Co-authored-by: Daniel Han <danielhanchen@gmail.com> Co-authored-by: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Co-authored-by: oobabooga <112222186+oobabooga@users.noreply.github.com>
This commit is contained in:
parent
63d8da34d3
commit
47fa4ca6c1
18 changed files with 1141 additions and 89 deletions
12
.github/workflows/studio-backend-ci.yml
vendored
12
.github/workflows/studio-backend-ci.yml
vendored
|
|
@ -193,6 +193,7 @@ jobs:
|
|||
--ignore=tests/sh \
|
||||
--ignore=tests/studio/test_hardware_dispatch_matrix.py \
|
||||
--ignore=tests/studio/test_is_mlx_dispatch_gate.py \
|
||||
--ignore=tests/studio/test_xpu_spoof_pipeline.py \
|
||||
--ignore=tests/vllm_compat \
|
||||
--ignore=tests/version_compat \
|
||||
-m 'not server and not e2e' \
|
||||
|
|
@ -205,14 +206,15 @@ jobs:
|
|||
env:
|
||||
PYTHONPATH: ${{ github.workspace }}/studio
|
||||
UNSLOTH_COMPILE_DISABLE: '1'
|
||||
# These two files mutate hardware.py module globals at runtime
|
||||
# via the spoof fixtures, which leaks state into any other test
|
||||
# that imports hardware. Run them in their own pytest invocation
|
||||
# so the leak does not cross file boundaries.
|
||||
# These files mutate hardware.py module globals at runtime via the
|
||||
# spoof fixtures (CUDA/ROCm/XPU/MLX/CPU), which leaks state into any
|
||||
# other test that imports hardware. Run them in their own pytest
|
||||
# invocation so the leak does not cross file boundaries.
|
||||
run: |
|
||||
python -m pytest -q --tb=short \
|
||||
tests/studio/test_hardware_dispatch_matrix.py \
|
||||
tests/studio/test_is_mlx_dispatch_gate.py
|
||||
tests/studio/test_is_mlx_dispatch_gate.py \
|
||||
tests/studio/test_xpu_spoof_pipeline.py
|
||||
|
||||
- name: Shell installer tests
|
||||
# Subset that does not depend on a writable / pristine install.sh
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ from unsloth.chat_templates import get_chat_template
|
|||
from transformers import TextIteratorStreamer, TextStreamer
|
||||
from peft import PeftModel, PeftModelForCausalLM
|
||||
|
||||
import contextlib
|
||||
import json
|
||||
import sys
|
||||
import torch
|
||||
|
|
@ -1942,8 +1943,30 @@ class InferenceBackend:
|
|||
+ text
|
||||
+ "<|text_end|>\n<|audio_start|><|global_features_start|>\n"
|
||||
)
|
||||
|
||||
with torch.inference_mode():
|
||||
with torch.amp.autocast("cuda", dtype = model.dtype):
|
||||
# Derive the autocast device from the loaded model, not from the
|
||||
# global backend: a CPU-fallback DAC on an XPU/CUDA host must not
|
||||
# open a GPU autocast context around CPU tensors.
|
||||
device_type = (
|
||||
model.device.type
|
||||
if hasattr(model.device, "type")
|
||||
else str(model.device).split(":", 1)[0]
|
||||
)
|
||||
# Clamp to autocast-supported backends so exotic devices
|
||||
# (e.g. "meta" during accelerate offloaded loading) do not raise.
|
||||
# MPS is autocast-supported since torch 2.3, keep it in the set.
|
||||
if device_type not in ("cuda", "xpu", "mps", "cpu"):
|
||||
device_type = "cpu"
|
||||
# CPU and XPU autocast only accept bfloat16/float16. For a
|
||||
# float32 model, skip autocast entirely to avoid raising or
|
||||
# producing a warning on every generate call.
|
||||
autocast_dtype_supported = model.dtype in (torch.bfloat16, torch.float16)
|
||||
if device_type in ("cpu", "xpu") and not autocast_dtype_supported:
|
||||
autocast_ctx = contextlib.nullcontext()
|
||||
else:
|
||||
autocast_ctx = torch.amp.autocast(device_type, dtype = model.dtype)
|
||||
with autocast_ctx:
|
||||
inputs = tokenizer([prompt], return_tensors = "pt").to(model.device)
|
||||
generated = model.generate(
|
||||
**inputs,
|
||||
|
|
|
|||
|
|
@ -27,7 +27,7 @@ import uuid
|
|||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from typing import Any, Generator, Optional, Tuple, Union
|
||||
from utils.hardware import prepare_gpu_selection
|
||||
from utils.hardware import get_device, prepare_gpu_selection
|
||||
|
||||
# Re-exported from the shared helper so GGUF, training, and inference share one
|
||||
# type; kept importable here for backwards compatibility.
|
||||
|
|
@ -1012,6 +1012,8 @@ class InferenceOrchestrator:
|
|||
)
|
||||
sub_config["resolved_gpu_ids"] = resolved_gpu_ids
|
||||
sub_config["gpu_selection"] = gpu_selection
|
||||
# Parent-detected backend for the worker's apply_gpu_ids().
|
||||
sub_config["device_backend"] = get_device().value
|
||||
|
||||
# Recheck the sidecar reservation BEFORE tearing the old worker down,
|
||||
# for REPAIRS only: an install holds this same lifecycle gate, so it
|
||||
|
|
|
|||
|
|
@ -794,7 +794,7 @@ def run_inference_process(
|
|||
env = os.getenv("ENVIRONMENT_TYPE", "production"),
|
||||
)
|
||||
|
||||
apply_gpu_ids(config.get("resolved_gpu_ids"))
|
||||
apply_gpu_ids(config.get("resolved_gpu_ids"), backend = config.get("device_backend"))
|
||||
|
||||
model_name = config["model_name"]
|
||||
|
||||
|
|
|
|||
|
|
@ -1481,6 +1481,9 @@ class UnslothTrainer:
|
|||
|
||||
SNAC_MODEL_NAME = "hubertsiuzdak/snac_24khz"
|
||||
SNAC_SAMPLE_RATE = 24000
|
||||
|
||||
# SNAC codec unvalidated on Intel XPU; keep the pre-PR CPU
|
||||
# fallback for non-CUDA hosts.
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
max_length = self.max_seq_length or 2048
|
||||
tokenizer = self.tokenizer
|
||||
|
|
@ -1642,7 +1645,8 @@ class UnslothTrainer:
|
|||
del snac_model
|
||||
|
||||
gc.collect()
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
clear_gpu_cache()
|
||||
self._cuda_audio_used = True
|
||||
|
||||
if not processed_examples:
|
||||
|
|
@ -1669,6 +1673,8 @@ class UnslothTrainer:
|
|||
import numpy as np
|
||||
import torchaudio.transforms as T
|
||||
|
||||
# Spark-TTS BiCodec unvalidated on Intel XPU; keep the pre-PR CPU
|
||||
# fallback for non-CUDA hosts.
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
|
||||
# sparktts lives in the SparkAudio/Spark-TTS GitHub repo, not the HF model
|
||||
|
|
@ -1857,7 +1863,8 @@ class UnslothTrainer:
|
|||
del audio_tokenizer
|
||||
|
||||
gc.collect()
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
clear_gpu_cache()
|
||||
self._cuda_audio_used = True
|
||||
|
||||
if not processed_examples:
|
||||
|
|
@ -1894,6 +1901,8 @@ class UnslothTrainer:
|
|||
from datasets import Dataset as HFDataset
|
||||
from utils.paths import ensure_dir, tmp_root
|
||||
|
||||
# OuteTTS DAC/Whisper preprocess unvalidated on Intel XPU; keep the
|
||||
# pre-PR CPU fallback for non-CUDA hosts.
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
|
||||
# Clone OuteTTS repo (same as audio_codecs._load_dac)
|
||||
|
|
@ -2065,7 +2074,8 @@ class UnslothTrainer:
|
|||
del prompt_processor
|
||||
|
||||
gc.collect()
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
clear_gpu_cache()
|
||||
self._cuda_audio_used = True
|
||||
|
||||
if not processed_examples:
|
||||
|
|
|
|||
|
|
@ -30,7 +30,7 @@ from typing import Optional, Tuple, Any, Callable, Union, TYPE_CHECKING
|
|||
|
||||
if TYPE_CHECKING:
|
||||
import matplotlib.pyplot as plt
|
||||
from utils.hardware import prepare_gpu_selection
|
||||
from utils.hardware import get_device, prepare_gpu_selection
|
||||
from utils.native_path_leases import (
|
||||
native_path_secret_removed_for_child_start,
|
||||
run_without_native_path_secret,
|
||||
|
|
@ -219,6 +219,9 @@ def _build_training_worker_config(values: dict[str, Any]) -> dict[str, Any]:
|
|||
config[key] = values.get(key)
|
||||
if config["training_type"] == "Full Finetuning":
|
||||
config["load_in_4bit"] = False
|
||||
# The parent's detected backend: the worker's apply_gpu_ids() targets the
|
||||
# right visibility env var from this, without probing torch pre-mask.
|
||||
config["device_backend"] = get_device().value
|
||||
return config
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -2373,7 +2373,7 @@ def run_training_process(*, event_queue: Any, stop_queue: Any, config: dict) ->
|
|||
env = os.getenv("ENVIRONMENT_TYPE", "production"),
|
||||
)
|
||||
|
||||
apply_gpu_ids(config.get("resolved_gpu_ids"))
|
||||
apply_gpu_ids(config.get("resolved_gpu_ids"), backend = config.get("device_backend"))
|
||||
|
||||
model_name = config["model_name"]
|
||||
|
||||
|
|
|
|||
|
|
@ -74,7 +74,16 @@ class LoadRequest(BaseModel):
|
|||
)
|
||||
gpu_ids: Optional[List[int]] = Field(
|
||||
None,
|
||||
description = "GPU placement pool, for example [0, 1]. Omit or pass [] to use automatic selection. CUDA/ROCm values are physical GPU indices and are unsupported when the parent CUDA_VISIBLE_DEVICES uses UUID/MIG entries; Vulkan values are ggml device ordinals. For GGUF models the fitter may pin the smallest subset of this pool that fits.",
|
||||
description = (
|
||||
"GPU placement pool, for example [0, 1]. Omit or pass [] to use "
|
||||
"automatic selection. CUDA/ROCm and Intel XPU values are physical "
|
||||
"GPU indices; Vulkan values are ggml device ordinals. Explicit "
|
||||
"physical IDs are unsupported when the parent visibility mask uses "
|
||||
"non-numeric or subdevice entries, including CUDA_VISIBLE_DEVICES "
|
||||
"with UUID/MIG entries and ZE_AFFINITY_MASK with subdevice tokens "
|
||||
"(for example '0.0,0.1') or FLAT-hierarchy tile handles. For GGUF "
|
||||
"models the fitter may pin the smallest subset of this pool that fits."
|
||||
),
|
||||
)
|
||||
speculative_type: Optional[str] = Field(
|
||||
None,
|
||||
|
|
|
|||
|
|
@ -496,7 +496,15 @@ class TrainingStartRequest(BaseModel):
|
|||
# GPU selection
|
||||
gpu_ids: Optional[List[int]] = Field(
|
||||
None,
|
||||
description = "Physical GPU indices to use, for example [0, 1]. Omit or pass [] to use automatic selection. Explicit gpu_ids are unsupported when the parent CUDA_VISIBLE_DEVICES uses UUID/MIG entries.",
|
||||
description = (
|
||||
"Physical GPU indices to use, for example [0, 1]. Omit or pass "
|
||||
"[] to use automatic selection. Explicit gpu_ids are unsupported "
|
||||
"when the parent visibility mask uses non-numeric or subdevice "
|
||||
"entries -- this includes CUDA_VISIBLE_DEVICES with UUID/MIG "
|
||||
"entries on NVIDIA, and ZE_AFFINITY_MASK with subdevice tokens "
|
||||
"(e.g. '0.0,0.1') or FLAT-hierarchy (default) tile handles on "
|
||||
"Intel XPU."
|
||||
),
|
||||
)
|
||||
|
||||
# S3 dataset source configuration
|
||||
|
|
|
|||
|
|
@ -135,8 +135,8 @@ def can_keep_chat_during_training(
|
|||
resolve_requested_gpu_ids,
|
||||
)
|
||||
|
||||
if get_device() != DeviceType.CUDA:
|
||||
return False, {"mode": "non_cuda", "reason": "non_cuda"}
|
||||
if get_device() not in (DeviceType.CUDA, DeviceType.XPU):
|
||||
return False, {"mode": "non_accelerator", "reason": "non_accelerator"}
|
||||
|
||||
# Full finetuning runs in 16-bit, so ignore the 4-bit request or we under-count.
|
||||
effective_4bit = False if training_type == "Full Finetuning" else load_in_4bit
|
||||
|
|
@ -241,8 +241,8 @@ def can_load_chat_during_training(
|
|||
N visible GPUs instead.
|
||||
``single_device_gpu`` is the exact physical device token selected by a
|
||||
single-device runner. `load_in_4bit` must be effective (LoRA can flip 4-bit
|
||||
-> 16-bit). Non-CUDA allows the load; default-deny on any CUDA case it can't
|
||||
size, so a load never OOMs training."""
|
||||
-> 16-bit). CPU/MLX allows the load; default-deny on any CUDA/XPU case it
|
||||
can't size, so a load never OOMs training."""
|
||||
try:
|
||||
from utils.hardware import (
|
||||
DeviceType,
|
||||
|
|
@ -253,8 +253,8 @@ def can_load_chat_during_training(
|
|||
resolve_requested_gpu_ids,
|
||||
)
|
||||
|
||||
if get_device() != DeviceType.CUDA:
|
||||
return True, {"mode": "non_cuda", "reason": "non_cuda"}
|
||||
if get_device() not in (DeviceType.CUDA, DeviceType.XPU):
|
||||
return True, {"mode": "non_accelerator", "reason": "non_accelerator"}
|
||||
|
||||
est_kwargs = dict(
|
||||
hf_token = hf_token or None,
|
||||
|
|
|
|||
|
|
@ -326,7 +326,7 @@ class TestCanLoadGGUF(_GpuCacheResetMixin, unittest.TestCase):
|
|||
|
||||
|
||||
class TestCanLoadMisc(_GpuCacheResetMixin, unittest.TestCase):
|
||||
def test_non_cuda_allows(self):
|
||||
def test_non_accelerator_allows(self):
|
||||
with patch("utils.hardware.get_device", return_value = DeviceType.MLX):
|
||||
ok, info = tv.can_load_chat_during_training(
|
||||
model_name = "m",
|
||||
|
|
@ -336,7 +336,30 @@ class TestCanLoadMisc(_GpuCacheResetMixin, unittest.TestCase):
|
|||
requested_gpu_ids = None,
|
||||
)
|
||||
self.assertTrue(ok)
|
||||
self.assertEqual(info["mode"], "non_cuda")
|
||||
self.assertEqual(info["mode"], "non_accelerator")
|
||||
|
||||
def test_xpu_overcommit_is_refused(self):
|
||||
# XPU must NOT get the blanket non-accelerator allow: an oversized
|
||||
# chat model during resident training is refused, like CUDA.
|
||||
with (
|
||||
patch("utils.hardware.get_device", return_value = DeviceType.XPU),
|
||||
patch(
|
||||
"utils.hardware.auto_select_gpu_ids",
|
||||
return_value = (
|
||||
None,
|
||||
{"selection_mode": "auto", "required_gb": 50.0, "usable_gb": 4.0},
|
||||
),
|
||||
),
|
||||
):
|
||||
ok, info = tv.can_load_chat_during_training(
|
||||
model_name = "m",
|
||||
hf_token = None,
|
||||
load_in_4bit = True,
|
||||
max_seq_length = 0,
|
||||
requested_gpu_ids = None,
|
||||
)
|
||||
self.assertFalse(ok)
|
||||
self.assertNotEqual(info.get("mode"), "non_accelerator")
|
||||
|
||||
def test_no_visible_gpus_refuses(self):
|
||||
# GGUF with an empty device list -> no candidate GPU -> default-deny.
|
||||
|
|
|
|||
|
|
@ -119,7 +119,8 @@ class TestResolveRequestedGpuIds(_GpuCacheResetMixin, unittest.TestCase):
|
|||
patch("utils.hardware.hardware.get_physical_gpu_count", return_value = 8),
|
||||
):
|
||||
with self.assertRaisesRegex(
|
||||
ValueError, "unsupported when CUDA_VISIBLE_DEVICES uses UUID/MIG"
|
||||
ValueError,
|
||||
"unsupported when CUDA_VISIBLE_DEVICES uses non-numeric or subdevice",
|
||||
):
|
||||
resolve_requested_gpu_ids([1])
|
||||
|
||||
|
|
@ -866,12 +867,12 @@ class TestPreSpawnGpuResolution(_GpuCacheResetMixin, unittest.TestCase):
|
|||
|
||||
|
||||
class TestRouteErrors(unittest.TestCase):
|
||||
def test_prepare_gpu_selection_rejects_gpu_ids_on_non_cuda_backend(self):
|
||||
def test_prepare_gpu_selection_rejects_gpu_ids_on_non_accelerator_backend(self):
|
||||
with patch("utils.hardware.hardware.get_device", return_value = DeviceType.CPU):
|
||||
with self.assertRaises(ValueError) as exc_info:
|
||||
prepare_gpu_selection([0], model_name = "unsloth/test")
|
||||
|
||||
self.assertIn("only supported on CUDA devices", str(exc_info.exception))
|
||||
self.assertIn("only supported on CUDA and Intel XPU", str(exc_info.exception))
|
||||
|
||||
def test_inference_route_resolves_gguf_gpu_ids(self):
|
||||
# GGUF gpu_ids are now supported: /load routes them through the same
|
||||
|
|
@ -1630,18 +1631,61 @@ class TestAutoSelectWithNoneRequired(_GpuCacheResetMixin, unittest.TestCase):
|
|||
self.assertEqual(metadata["selection_mode"], "fallback_all")
|
||||
|
||||
|
||||
class TestXpuRejection(_GpuCacheResetMixin, unittest.TestCase):
|
||||
def test_auto_select_returns_non_cuda_for_xpu(self):
|
||||
with patch("utils.hardware.hardware.get_device", return_value = DeviceType.XPU):
|
||||
class TestXpuSelection(_GpuCacheResetMixin, unittest.TestCase):
|
||||
def test_auto_select_supports_xpu(self):
|
||||
with (
|
||||
patch("utils.hardware.hardware.get_device", return_value = DeviceType.XPU),
|
||||
patch(
|
||||
"utils.hardware.hardware.estimate_required_model_memory_gb",
|
||||
return_value = (1.0, {}),
|
||||
),
|
||||
patch(
|
||||
"utils.hardware.hardware.get_visible_gpu_utilization",
|
||||
return_value = {
|
||||
"devices": [
|
||||
{"index": 0, "vram_total_gb": 8, "vram_used_gb": 1},
|
||||
]
|
||||
},
|
||||
),
|
||||
patch(
|
||||
"utils.hardware.hardware._get_parent_visible_gpu_spec",
|
||||
return_value = {
|
||||
"raw": None,
|
||||
"numeric_ids": [0],
|
||||
"supports_explicit_gpu_ids": True,
|
||||
},
|
||||
),
|
||||
patch(
|
||||
"utils.hardware.hardware.get_parent_visible_gpu_ids",
|
||||
return_value = [0],
|
||||
),
|
||||
):
|
||||
selected, metadata = auto_select_gpu_ids("unsloth/test")
|
||||
|
||||
self.assertIsNone(selected)
|
||||
self.assertEqual(metadata["selection_mode"], "non_cuda")
|
||||
self.assertEqual(selected, [0])
|
||||
self.assertEqual(metadata["selection_mode"], "auto")
|
||||
|
||||
def test_prepare_gpu_selection_rejects_explicit_ids_on_xpu(self):
|
||||
with patch("utils.hardware.hardware.get_device", return_value = DeviceType.XPU):
|
||||
with self.assertRaisesRegex(ValueError, "only supported on CUDA"):
|
||||
prepare_gpu_selection([0], model_name = "unsloth/test")
|
||||
def test_prepare_gpu_selection_accepts_explicit_ids_on_xpu(self):
|
||||
with (
|
||||
patch("utils.hardware.hardware.get_device", return_value = DeviceType.XPU),
|
||||
patch(
|
||||
"utils.hardware.hardware._get_parent_visible_gpu_spec",
|
||||
return_value = {
|
||||
"raw": "0",
|
||||
"numeric_ids": [0],
|
||||
"supports_explicit_gpu_ids": True,
|
||||
},
|
||||
),
|
||||
patch(
|
||||
"utils.hardware.hardware.get_parent_visible_gpu_ids",
|
||||
return_value = [0],
|
||||
),
|
||||
patch("utils.hardware.hardware.get_physical_gpu_count", return_value = 1),
|
||||
):
|
||||
selected, metadata = prepare_gpu_selection([0], model_name = "unsloth/test")
|
||||
|
||||
self.assertEqual(selected, [0])
|
||||
self.assertEqual(metadata["selection_mode"], "explicit")
|
||||
|
||||
|
||||
class TestEstimateFp16ModelSizeBytesPrefersLocalWeights(unittest.TestCase):
|
||||
|
|
|
|||
|
|
@ -294,13 +294,13 @@ class TestAutoSelectGpuIds(unittest.TestCase):
|
|||
# 35GB (first) + 30*0.85 (second) = 60.5GB > 50GB
|
||||
self.assertEqual(len(selected), 2)
|
||||
|
||||
def test_non_cuda_returns_none(self):
|
||||
def test_non_accelerator_returns_none(self):
|
||||
from utils.hardware.hardware import auto_select_gpu_ids
|
||||
import utils.hardware.hardware as hw
|
||||
with patch.object(hw, "get_device", return_value = hw.DeviceType.CPU):
|
||||
selected, meta = auto_select_gpu_ids("test/model")
|
||||
self.assertIsNone(selected)
|
||||
self.assertEqual(meta["selection_mode"], "non_cuda")
|
||||
self.assertEqual(meta["selection_mode"], "non_accelerator")
|
||||
|
||||
|
||||
class TestGetDeviceMap(unittest.TestCase):
|
||||
|
|
|
|||
|
|
@ -326,12 +326,21 @@ class TestCanKeepAuto(_GpuCacheResetMixin, unittest.TestCase):
|
|||
keep, _, _ = self._run((None, meta))
|
||||
self.assertFalse(keep)
|
||||
|
||||
def test_unload_on_non_cuda(self):
|
||||
def test_unload_on_non_accelerator(self):
|
||||
keep, info, auto_mock = self._run(([0], {}), device = DeviceType.CPU)
|
||||
self.assertFalse(keep)
|
||||
self.assertEqual(info["mode"], "non_cuda")
|
||||
self.assertEqual(info["mode"], "non_accelerator")
|
||||
auto_mock.assert_not_called()
|
||||
|
||||
def test_xpu_gets_sized_like_cuda(self):
|
||||
# XPU is a first-class training backend: the keep-guard must size it,
|
||||
# not blanket-unload it as a non-accelerator.
|
||||
meta = {"selection_mode": "auto", "required_gb": 10.0, "usable_gb": 30.0}
|
||||
keep, info, auto_mock = self._run(([0], meta), device = DeviceType.XPU)
|
||||
self.assertTrue(keep)
|
||||
self.assertNotEqual(info.get("mode"), "non_accelerator")
|
||||
auto_mock.assert_called_once()
|
||||
|
||||
def test_full_finetuning_forces_16bit_in_estimate(self):
|
||||
meta = {"selection_mode": "auto", "required_gb": 10.0, "usable_gb": 30.0}
|
||||
_keep, _info, auto_mock = self._run(
|
||||
|
|
|
|||
|
|
@ -50,6 +50,11 @@ def export_capability() -> dict:
|
|||
return _hardware.export_capability()
|
||||
|
||||
|
||||
def get_torch_device_str() -> str:
|
||||
"""Return the torch device string ("cuda", "xpu", "cpu") for the detected hardware."""
|
||||
return _hardware.get_torch_device_str()
|
||||
|
||||
|
||||
__all__ = [
|
||||
"DeviceType",
|
||||
"DEVICE",
|
||||
|
|
@ -75,6 +80,7 @@ __all__ = [
|
|||
"estimate_required_model_memory_gb",
|
||||
"auto_select_gpu_ids",
|
||||
"prepare_gpu_selection",
|
||||
"get_torch_device_str",
|
||||
"safe_num_proc",
|
||||
"safe_thread_num_proc",
|
||||
"dataset_map_num_proc",
|
||||
|
|
|
|||
|
|
@ -175,18 +175,64 @@ def detect_hardware() -> DeviceType:
|
|||
Call once at FastAPI lifespan startup; idempotent.
|
||||
|
||||
Detection order:
|
||||
1. CUDA (NVIDIA GPU, requires torch)
|
||||
2. MLX (Apple Silicon via MLX framework)
|
||||
3. CPU (fallback)
|
||||
1. XPU-preferred hint: only on an unambiguous "prefer XPU" signal
|
||||
(CUDA hidden via ``CUDA_VISIBLE_DEVICES="" / "-1"``,
|
||||
``UNSLOTH_FORCE_XPU=1``, or CUDA unavailable) AND a non-empty
|
||||
``ZE_AFFINITY_MASK`` AND ``torch.xpu`` reports a device. A stray
|
||||
inherited mask is not enough: CUDA still wins on hybrid hosts.
|
||||
2. CUDA (NVIDIA GPU, requires torch)
|
||||
3. XPU (Intel GPU, requires torch with XPU support)
|
||||
4. MLX (Apple Silicon via MLX framework)
|
||||
5. CPU (fallback)
|
||||
"""
|
||||
global DEVICE, CHAT_ONLY, CHAT_ONLY_REASON, IS_ROCM
|
||||
CHAT_ONLY = True # reset -- only CUDA/ROCm/XPU/MLX sets it to False
|
||||
CHAT_ONLY_REASON = None
|
||||
IS_ROCM = False
|
||||
|
||||
# --- CUDA / ROCm: try PyTorch ---
|
||||
# --- CUDA / ROCm / XPU: try PyTorch ---
|
||||
if _has_torch():
|
||||
import torch
|
||||
|
||||
# --- Explicit-XPU hint ---
|
||||
# Prefer XPU on UNSLOTH_FORCE_XPU=1, or ZE_AFFINITY_MASK set + CUDA
|
||||
# hidden/unavailable. A bare mask alone is NOT enough (can leak from
|
||||
# unrelated Intel tooling); torch.xpu must report a device.
|
||||
ze_mask = os.environ.get("ZE_AFFINITY_MASK")
|
||||
cvd = os.environ.get("CUDA_VISIBLE_DEVICES")
|
||||
cuda_hidden = cvd is not None and cvd.strip() in ("", "-1")
|
||||
force_xpu = os.environ.get("UNSLOTH_FORCE_XPU") == "1"
|
||||
try:
|
||||
cuda_unavailable = not torch.cuda.is_available()
|
||||
except Exception:
|
||||
cuda_unavailable = True
|
||||
|
||||
prefer_xpu = force_xpu or (bool(ze_mask) and (cuda_hidden or cuda_unavailable))
|
||||
if prefer_xpu:
|
||||
try:
|
||||
xpu_ok = hasattr(torch, "xpu") and torch.xpu.is_available()
|
||||
except Exception:
|
||||
xpu_ok = False
|
||||
if xpu_ok:
|
||||
# Forced XPU on a hybrid host: unsloth's device_type picks
|
||||
# CUDA before XPU and ignores this Studio-only env var, so
|
||||
# hide CUDA or spawned workers would silently train on CUDA.
|
||||
if force_xpu and not cuda_hidden and not cuda_unavailable:
|
||||
os.environ["CUDA_VISIBLE_DEVICES"] = ""
|
||||
DEVICE = DeviceType.XPU
|
||||
CHAT_ONLY = False
|
||||
CHAT_ONLY_REASON = None
|
||||
device_name = torch.xpu.get_device_name(0)
|
||||
if force_xpu and not ze_mask:
|
||||
reason = "UNSLOTH_FORCE_XPU=1"
|
||||
elif force_xpu:
|
||||
reason = "UNSLOTH_FORCE_XPU=1 + ZE_AFFINITY_MASK"
|
||||
else:
|
||||
reason = "ZE_AFFINITY_MASK hint honoured"
|
||||
print(f"Hardware detected: XPU -- {device_name} ({reason})")
|
||||
return DEVICE
|
||||
|
||||
# --- CUDA: NVIDIA GPU ---
|
||||
if torch.cuda.is_available():
|
||||
DEVICE = DeviceType.CUDA
|
||||
CHAT_ONLY = False
|
||||
|
|
@ -327,9 +373,18 @@ def clear_gpu_cache():
|
|||
torch.cuda.empty_cache()
|
||||
torch.cuda.ipc_collect()
|
||||
elif device == DeviceType.XPU:
|
||||
import torch
|
||||
torch.xpu.synchronize()
|
||||
torch.xpu.empty_cache()
|
||||
# Guard synchronize/empty_cache: older torch-xpu builds may lack
|
||||
# them, and an unguarded AttributeError would propagate to callers.
|
||||
# torch.xpu has no ipc_collect(), so do not call it here.
|
||||
try:
|
||||
import torch
|
||||
if hasattr(torch, "xpu"):
|
||||
if hasattr(torch.xpu, "synchronize"):
|
||||
torch.xpu.synchronize()
|
||||
if hasattr(torch.xpu, "empty_cache"):
|
||||
torch.xpu.empty_cache()
|
||||
except Exception as e:
|
||||
logger.debug("Failed to clear XPU cache: %s", e)
|
||||
elif device == DeviceType.MLX:
|
||||
# MLX manages memory automatically; gc.collect() above is enough.
|
||||
pass
|
||||
|
|
@ -500,14 +555,27 @@ def get_package_versions() -> Dict[str, Optional[str]]:
|
|||
except PackageNotFoundError:
|
||||
versions[name] = None
|
||||
|
||||
# GPU runtime version bundled with torch
|
||||
# GPU runtime versions bundled with torch (CUDA, ROCm/HIP, Intel XPU)
|
||||
try:
|
||||
import torch
|
||||
|
||||
versions["cuda"] = getattr(torch.version, "cuda", None)
|
||||
versions["rocm"] = getattr(torch.version, "hip", None)
|
||||
# Isolated probe: a broken Intel runtime raising in is_available()
|
||||
# must not blank the already-read cuda/rocm versions.
|
||||
try:
|
||||
if hasattr(torch, "xpu") and torch.xpu.is_available():
|
||||
# torch.version.xpu may be None on modern builds; fall back to
|
||||
# "available" so the UI distinguishes present-but-unknown from
|
||||
# "package not found".
|
||||
xpu_ver = getattr(torch.version, "xpu", None)
|
||||
versions["xpu"] = xpu_ver if xpu_ver is not None else "available"
|
||||
except Exception:
|
||||
versions["xpu"] = None
|
||||
except Exception:
|
||||
versions["cuda"] = None
|
||||
versions["rocm"] = None
|
||||
versions["xpu"] = None
|
||||
|
||||
return versions
|
||||
|
||||
|
|
@ -547,6 +615,7 @@ def _torch_get_per_device_info(device_indices: list[int]) -> list[Dict[str, Any]
|
|||
if mod is None:
|
||||
return []
|
||||
|
||||
device = get_device()
|
||||
# free==total is a Windows-ROCm-only quirk.
|
||||
_win_rocm = sys.platform == "win32" and IS_ROCM
|
||||
devices = []
|
||||
|
|
@ -558,11 +627,30 @@ def _torch_get_per_device_info(device_indices: list[int]) -> list[Dict[str, Any]
|
|||
used_bytes: Optional[int]
|
||||
# Prefer mem_get_info (system-wide) so auto-select sees other consumers.
|
||||
if hasattr(mod, "mem_get_info"):
|
||||
free_bytes, total_bytes = mod.mem_get_info(ordinal)
|
||||
used_bytes = total_bytes - free_bytes
|
||||
# free==total is the broken-API sentinel, not an idle GPU.
|
||||
if _win_rocm and free_bytes == total_bytes:
|
||||
try:
|
||||
free_bytes, total_bytes = mod.mem_get_info(ordinal)
|
||||
used_bytes = total_bytes - free_bytes
|
||||
except Exception as e:
|
||||
if device != DeviceType.XPU:
|
||||
raise
|
||||
# Arc B580 and Lunar Lake can report properties while
|
||||
# rejecting free-memory queries. Preserve the usable
|
||||
# device and its total memory with unknown utilization.
|
||||
logger.debug(
|
||||
"XPU free-memory query failed for ordinal %d: %s",
|
||||
ordinal,
|
||||
e,
|
||||
)
|
||||
used_bytes = None
|
||||
else:
|
||||
# free==total is the broken-API sentinel, not an idle GPU.
|
||||
if _win_rocm and free_bytes == total_bytes:
|
||||
used_bytes = None
|
||||
elif device == DeviceType.XPU:
|
||||
# XPU without mem_get_info: memory_allocated() is process-local
|
||||
# and misleading for placement, so return None for the
|
||||
# selector's no-telemetry fallback.
|
||||
used_bytes = None
|
||||
else:
|
||||
used_bytes = mod.memory_allocated(ordinal)
|
||||
devices.append(
|
||||
|
|
@ -571,7 +659,9 @@ def _torch_get_per_device_info(device_indices: list[int]) -> list[Dict[str, Any]
|
|||
"visible_ordinal": ordinal,
|
||||
"name": props.name,
|
||||
"total_gb": round(total_bytes / (1024**3), 2),
|
||||
"used_gb": round(used_bytes / (1024**3), 2) if used_bytes is not None else None,
|
||||
"used_gb": (
|
||||
round(used_bytes / (1024**3), 2) if used_bytes is not None else None
|
||||
),
|
||||
}
|
||||
)
|
||||
except Exception as e:
|
||||
|
|
@ -582,6 +672,43 @@ def _torch_get_per_device_info(device_indices: list[int]) -> list[Dict[str, Any]
|
|||
# ========== Live GPU Utilization ==========
|
||||
|
||||
|
||||
def _xpu_hierarchy_is_composite() -> bool:
|
||||
"""Return True iff Level Zero is running in COMPOSITE device hierarchy.
|
||||
|
||||
COMPOSITE: numeric ``ZE_AFFINITY_MASK`` entries address root GPU IDs
|
||||
(tiles use ``N.M``). FLAT (the oneAPI default; also assumed when
|
||||
``ZE_FLAT_DEVICE_HIERARCHY`` is unset): entries address tile/device
|
||||
handles, so mapping them back to root GPU IDs is unsafe. Only COMPOSITE
|
||||
gives stable root-ID semantics.
|
||||
"""
|
||||
hierarchy = (os.environ.get("ZE_FLAT_DEVICE_HIERARCHY") or "FLAT").strip().upper()
|
||||
return hierarchy == "COMPOSITE"
|
||||
|
||||
|
||||
def _parse_ze_mask_roots(mask: str) -> list[int]:
|
||||
"""Parse a ``ZE_AFFINITY_MASK`` value into an ordered list of root device IDs.
|
||||
|
||||
One root ID per mask token, preserving order and duplicates so logical
|
||||
ordinals map 1-to-1 to physical root IDs (e.g. ``"0.0,0.1"`` -> ``[0, 0]``,
|
||||
``"2.0,0.1,0.2"`` -> ``[2, 0, 0]``); empty list if no parseable digits.
|
||||
Only meaningful in COMPOSITE hierarchy -- callers needing a stable
|
||||
root-ID mapping must gate on ``_xpu_hierarchy_is_composite()``.
|
||||
"""
|
||||
roots: list[int] = []
|
||||
if not mask:
|
||||
return roots
|
||||
for token in mask.split(","):
|
||||
token = token.strip()
|
||||
if not token:
|
||||
continue
|
||||
root = token.split(".", 1)[0]
|
||||
# isdecimal() (not isdigit()) rejects Unicode superscripts like
|
||||
# "²"/"³", which pass isdigit() but crash int() with ValueError.
|
||||
if root.isdecimal():
|
||||
roots.append(int(root))
|
||||
return roots
|
||||
|
||||
|
||||
def _smi_query(func_name: str, *args, **kwargs) -> Optional[Dict[str, Any]]:
|
||||
"""Query the appropriate SMI backend (amd-smi or nvidia-smi).
|
||||
|
||||
|
|
@ -1504,6 +1631,13 @@ def get_visible_gpu_utilization() -> Dict[str, Any]:
|
|||
for td in torch_devices:
|
||||
total = td["total_gb"]
|
||||
used = td["used_gb"]
|
||||
# used=None is a deliberate "telemetry unavailable" signal
|
||||
# from _torch_get_per_device_info (e.g. XPU without
|
||||
# mem_get_info); propagate None instead of dividing by it. On
|
||||
# CUDA/ROCm used is always an int, so this stays byte-identical.
|
||||
vram_pct = (
|
||||
round((used / total) * 100, 1) if used is not None and total > 0 else None
|
||||
)
|
||||
devices.append(
|
||||
{
|
||||
"index": td["index"],
|
||||
|
|
@ -1513,9 +1647,7 @@ def get_visible_gpu_utilization() -> Dict[str, Any]:
|
|||
"temperature_c": None,
|
||||
"vram_used_gb": used,
|
||||
"vram_total_gb": total,
|
||||
"vram_utilization_pct": round((used / total) * 100, 1)
|
||||
if total > 0 and used is not None
|
||||
else None,
|
||||
"vram_utilization_pct": vram_pct,
|
||||
"power_draw_w": None,
|
||||
"power_limit_w": None,
|
||||
"power_utilization_pct": None,
|
||||
|
|
@ -1583,6 +1715,82 @@ _visible_gpu_count: Optional[int] = None
|
|||
|
||||
|
||||
def _get_parent_visible_gpu_spec() -> Dict[str, Any]:
|
||||
# On Intel XPU, visibility is controlled by ZE_AFFINITY_MASK (Level Zero),
|
||||
# not CUDA_VISIBLE_DEVICES.
|
||||
if get_device() == DeviceType.XPU:
|
||||
xpu_mask_raw = os.environ.get("ZE_AFFINITY_MASK")
|
||||
composite = _xpu_hierarchy_is_composite()
|
||||
|
||||
if xpu_mask_raw is None:
|
||||
# COMPOSITE: root GPU IDs are stable physical IDs.
|
||||
if composite:
|
||||
return {
|
||||
"raw": None,
|
||||
"numeric_ids": list(range(get_physical_gpu_count())),
|
||||
"supports_explicit_gpu_ids": True,
|
||||
}
|
||||
# FLAT (oneAPI default): ordinals are tile/device handles, not
|
||||
# physical GPU IDs. numeric_ids=None so telemetry uses relative
|
||||
# ordinals; explicit selection needs ZE_FLAT_DEVICE_HIERARCHY=COMPOSITE.
|
||||
return {
|
||||
"raw": None,
|
||||
"numeric_ids": None,
|
||||
"supports_explicit_gpu_ids": False,
|
||||
}
|
||||
|
||||
xpu_mask = xpu_mask_raw.strip()
|
||||
if xpu_mask == "":
|
||||
return {
|
||||
"raw": xpu_mask,
|
||||
"numeric_ids": [],
|
||||
"supports_explicit_gpu_ids": True,
|
||||
}
|
||||
|
||||
# Subdevice syntax ("N.M") expands one root into multiple
|
||||
# logical devices -- not addressable by explicit root-ID selection.
|
||||
has_subdevice = any("." in token.strip() for token in xpu_mask.split(",") if token.strip())
|
||||
if has_subdevice:
|
||||
return {
|
||||
"raw": xpu_mask,
|
||||
"numeric_ids": None,
|
||||
"supports_explicit_gpu_ids": False,
|
||||
}
|
||||
|
||||
# FLAT numeric entries are tile handles, not physical GPU IDs. Keep
|
||||
# numeric_ids unresolved so every telemetry and picker consumer uses
|
||||
# relative torch ordinals and cannot advertise them as pinnable roots.
|
||||
if not composite:
|
||||
tokens = [token.strip() for token in xpu_mask.split(",") if token.strip()]
|
||||
if tokens and all(token.isdecimal() for token in tokens):
|
||||
return {
|
||||
"raw": xpu_mask,
|
||||
"numeric_ids": None,
|
||||
"supports_explicit_gpu_ids": False,
|
||||
}
|
||||
return {
|
||||
"raw": xpu_mask,
|
||||
"numeric_ids": None,
|
||||
"supports_explicit_gpu_ids": False,
|
||||
}
|
||||
|
||||
# COMPOSITE + pure numeric (subdevice handled above). _parse_ze_mask_roots
|
||||
# maps to root GPU IDs, dropping non-decimal tokens so "*"/"GPU-uuid" -> [].
|
||||
roots_with_dupes = _parse_ze_mask_roots(xpu_mask)
|
||||
if not roots_with_dupes:
|
||||
# Unparseable mask (e.g. "*", "GPU-uuid") -- cannot map to
|
||||
# physical root IDs.
|
||||
return {
|
||||
"raw": xpu_mask,
|
||||
"numeric_ids": None,
|
||||
"supports_explicit_gpu_ids": False,
|
||||
}
|
||||
|
||||
return {
|
||||
"raw": xpu_mask,
|
||||
"numeric_ids": roots_with_dupes,
|
||||
"supports_explicit_gpu_ids": True,
|
||||
}
|
||||
|
||||
# ROCm uses HIP/ROCR_VISIBLE_DEVICES on top of CUDA_VISIBLE_DEVICES; check
|
||||
# them first. Explicit None checks (not `or`) so "" reads as "no visible GPUs".
|
||||
cuda_visible = None
|
||||
|
|
@ -1669,11 +1877,14 @@ def resolve_requested_gpu_ids(
|
|||
return requested_ids
|
||||
|
||||
if not parent_visible_spec["supports_explicit_gpu_ids"]:
|
||||
env_var_name = (
|
||||
"ZE_AFFINITY_MASK" if get_device() == DeviceType.XPU else "CUDA_VISIBLE_DEVICES"
|
||||
)
|
||||
raise ValueError(
|
||||
f"Invalid gpu_ids {requested_ids}: explicit physical GPU IDs are "
|
||||
f"unsupported when CUDA_VISIBLE_DEVICES uses UUID/MIG entries "
|
||||
f"({parent_visible_spec['raw']!r}). Omit gpu_ids to use the "
|
||||
"parent-visible devices."
|
||||
f"unsupported when {env_var_name} uses non-numeric or subdevice "
|
||||
f"entries ({parent_visible_spec['raw']!r}). Omit gpu_ids to use "
|
||||
"the parent-visible devices."
|
||||
)
|
||||
|
||||
if len(set(requested_ids)) != len(requested_ids):
|
||||
|
|
@ -2112,8 +2323,11 @@ def auto_select_gpu_ids(
|
|||
) -> tuple[Optional[list[int]], Dict[str, Any]]:
|
||||
metadata: Dict[str, Any] = {"selection_mode": "auto"}
|
||||
|
||||
if get_device() != DeviceType.CUDA:
|
||||
metadata["selection_mode"] = "non_cuda"
|
||||
# Auto-selection needs per-device free-VRAM telemetry, available on CUDA
|
||||
# (nvidia-smi) and XPU (torch.xpu) but not MLX/CPU, which fall
|
||||
# through to inheriting parent visibility.
|
||||
if get_device() not in (DeviceType.CUDA, DeviceType.XPU):
|
||||
metadata["selection_mode"] = "non_accelerator"
|
||||
return None, metadata
|
||||
|
||||
required_gb, estimate_metadata = estimate_required_model_memory_gb(
|
||||
|
|
@ -2272,10 +2486,10 @@ def prepare_gpu_selection(
|
|||
to a Hugging Face ``device_map`` string) and to ``apply_gpu_ids()`` in the
|
||||
worker subprocess (narrows ``CUDA_VISIBLE_DEVICES`` before torch/CUDA init).
|
||||
"""
|
||||
if gpu_ids and get_device() != DeviceType.CUDA:
|
||||
if gpu_ids and get_device() not in (DeviceType.CUDA, DeviceType.XPU):
|
||||
raise ValueError(
|
||||
f"gpu_ids {list(gpu_ids)} is only supported on CUDA devices, "
|
||||
f"but the current backend is '{get_device().value}'."
|
||||
f"gpu_ids {list(gpu_ids)} is only supported on CUDA and Intel XPU "
|
||||
f"devices, but the current backend is '{get_device().value}'."
|
||||
)
|
||||
|
||||
if gpu_ids:
|
||||
|
|
@ -2348,11 +2562,14 @@ def get_physical_gpu_count() -> int:
|
|||
def _backend_visible_devices_env() -> Optional[str]:
|
||||
"""Return the raw visibility env string that applies to this backend.
|
||||
|
||||
On ROCm, HIP_VISIBLE_DEVICES / ROCR_VISIBLE_DEVICES take precedence over
|
||||
CUDA_VISIBLE_DEVICES; this mirrors ``_get_parent_visible_gpu_spec`` so
|
||||
On XPU the control is ``ZE_AFFINITY_MASK`` (not ``CUDA_VISIBLE_DEVICES``);
|
||||
on ROCm, HIP_VISIBLE_DEVICES / ROCR_VISIBLE_DEVICES take precedence over
|
||||
CUDA_VISIBLE_DEVICES. Mirrors ``_get_parent_visible_gpu_spec`` so
|
||||
``backend_cuda_visible_devices`` reports the value actually narrowing the
|
||||
visible device set.
|
||||
visible device set on the current backend.
|
||||
"""
|
||||
if get_device() == DeviceType.XPU:
|
||||
return os.environ.get("ZE_AFFINITY_MASK")
|
||||
if IS_ROCM:
|
||||
return _get_parent_visible_gpu_spec().get("raw")
|
||||
return os.environ.get("CUDA_VISIBLE_DEVICES")
|
||||
|
|
@ -2467,6 +2684,43 @@ def get_visible_gpu_count() -> int:
|
|||
if _visible_gpu_count is not None:
|
||||
return _visible_gpu_count
|
||||
|
||||
# Prefer torch.xpu.device_count() on Intel XPU: the Level Zero runtime
|
||||
# correctly interprets ZE_AFFINITY_MASK semantics (e.g. subdevice syntax
|
||||
# "0.0,0.1" collapses onto one root GPU). Supersedes the torch fallback below.
|
||||
if get_device() == DeviceType.XPU:
|
||||
xpu_mask_raw = os.environ.get("ZE_AFFINITY_MASK")
|
||||
xpu_mask_set = xpu_mask_raw is not None
|
||||
xpu_visible = (xpu_mask_raw or "").strip()
|
||||
if xpu_mask_set and xpu_visible == "":
|
||||
_visible_gpu_count = 0
|
||||
return _visible_gpu_count
|
||||
|
||||
try:
|
||||
import torch
|
||||
_visible_gpu_count = torch.xpu.device_count()
|
||||
except Exception as e:
|
||||
logger.debug(
|
||||
"torch.xpu.device_count() failed, falling back to mask parsing: %s",
|
||||
e,
|
||||
)
|
||||
if xpu_visible:
|
||||
# Fallback: count unique root device IDs from the mask.
|
||||
# "device.subdevice" notation means "0.0,0.1" is 1 root, not 2.
|
||||
# Without torch the hierarchy mode is unknown, so root-device
|
||||
# counting is the conservative choice.
|
||||
if xpu_visible == "*":
|
||||
# Documented wildcard: all physical XPUs visible.
|
||||
_visible_gpu_count = get_physical_gpu_count()
|
||||
else:
|
||||
roots = _parse_ze_mask_roots(xpu_visible)
|
||||
# Non-parseable masks (",,,", "GPU-abc") yield an empty
|
||||
# roots list, treated as 0 visible devices, not "all
|
||||
# visible" -- no evidence the whole fleet was intended.
|
||||
_visible_gpu_count = len(set(roots))
|
||||
else:
|
||||
_visible_gpu_count = get_physical_gpu_count()
|
||||
return _visible_gpu_count
|
||||
|
||||
# _get_parent_visible_gpu_spec() already handles HIP_VISIBLE_DEVICES /
|
||||
# ROCR_VISIBLE_DEVICES on ROCm.
|
||||
visible_spec = _get_parent_visible_gpu_spec()
|
||||
|
|
@ -2480,20 +2734,18 @@ def get_visible_gpu_count() -> int:
|
|||
_visible_gpu_count = len([x for x in raw.split(",") if x.strip()])
|
||||
return _visible_gpu_count
|
||||
|
||||
# No visibility env var set -- try torch, else physical count
|
||||
# No visibility env var set -- try torch, else physical count. XPU is
|
||||
# handled by the early return above, so only torch.cuda is needed here.
|
||||
try:
|
||||
import torch
|
||||
if get_device() == DeviceType.XPU and hasattr(torch, "xpu"):
|
||||
_visible_gpu_count = torch.xpu.device_count()
|
||||
else:
|
||||
_visible_gpu_count = torch.cuda.device_count()
|
||||
_visible_gpu_count = torch.cuda.device_count()
|
||||
except Exception:
|
||||
_visible_gpu_count = get_physical_gpu_count()
|
||||
|
||||
return _visible_gpu_count
|
||||
|
||||
|
||||
def apply_gpu_ids(gpu_ids) -> None:
|
||||
def apply_gpu_ids(gpu_ids, backend: Optional[str] = None) -> None:
|
||||
if gpu_ids is None:
|
||||
return
|
||||
|
||||
|
|
@ -2509,6 +2761,62 @@ def apply_gpu_ids(gpu_ids) -> None:
|
|||
else:
|
||||
value = str(gpu_ids)
|
||||
|
||||
# Intel XPU honors ZE_AFFINITY_MASK, not CUDA_VISIBLE_DEVICES; route XPU
|
||||
# pinning through it so worker subprocesses are restricted to the intended GPU.
|
||||
# Decide WITHOUT get_device(): workers call this before detect_hardware(),
|
||||
# and a lazy detect would probe torch.cuda against the unmasked parent env,
|
||||
# latching device enumeration before the mask below is written. Pre-detect,
|
||||
# use env + torch BUILD attributes only (no runtime init, like the ROCm
|
||||
# mirror below).
|
||||
_is_xpu = DEVICE == DeviceType.XPU
|
||||
if backend is not None:
|
||||
# The spawning parent's detected backend (config["device_backend"]):
|
||||
# exact and probe-free, so the mask target always matches what
|
||||
# detect_hardware() decided in the parent, including its XPU
|
||||
# availability check and CUDA fallback.
|
||||
_is_xpu = backend == DeviceType.XPU.value
|
||||
elif DEVICE is None:
|
||||
# No parent backend passed (direct caller). version.xpu can be None
|
||||
# on a working XPU build, so also accept torch.xpu._is_compiled()
|
||||
# (a pure symbol-presence check, no runtime init). UNSLOTH_FORCE_XPU
|
||||
# counts only on an XPU-capable build: detect_hardware() falls back
|
||||
# to CUDA when XPU is missing, and the mask target must follow.
|
||||
try:
|
||||
import torch as _torch
|
||||
|
||||
_ver = _torch.version
|
||||
_is_comp = getattr(getattr(_torch, "xpu", None), "_is_compiled", None)
|
||||
_xpu_build = (callable(_is_comp) and bool(_is_comp())) or (
|
||||
getattr(_ver, "xpu", None) is not None
|
||||
)
|
||||
if os.environ.get("UNSLOTH_FORCE_XPU") == "1":
|
||||
_is_xpu = _xpu_build
|
||||
else:
|
||||
# Mirror detect_hardware: hidden CUDA prefers XPU on an
|
||||
# XPU-capable build (with or without a ZE mask -- detection
|
||||
# falls through to XPU either way), where writing these ids
|
||||
# to CUDA_VISIBLE_DEVICES would re-expose the deliberately
|
||||
# hidden CUDA.
|
||||
_cvd = os.environ.get("CUDA_VISIBLE_DEVICES")
|
||||
_cuda_hidden = _cvd is not None and _cvd.strip() in ("", "-1")
|
||||
_is_xpu = _xpu_build and (
|
||||
_cuda_hidden
|
||||
or (getattr(_ver, "cuda", None) is None and getattr(_ver, "hip", None) is None)
|
||||
)
|
||||
except Exception as e:
|
||||
logger.debug(
|
||||
"apply_gpu_ids: torch XPU probe skipped (%s: %s)",
|
||||
type(e).__name__,
|
||||
e,
|
||||
)
|
||||
if _is_xpu:
|
||||
os.environ["ZE_AFFINITY_MASK"] = value
|
||||
# Leave inherited CUDA_VISIBLE_DEVICES alone -- clearing it could let
|
||||
# the worker flip back to CUDA on hybrid hosts.
|
||||
_visible_gpu_count = None
|
||||
logger.info("Applied gpu_ids: ZE_AFFINITY_MASK='%s'", value)
|
||||
return
|
||||
|
||||
os.environ["CUDA_VISIBLE_DEVICES"] = value
|
||||
# Keep ROCm visibility env vars in sync. Workers may call apply_gpu_ids()
|
||||
# before detect_hardware() (IS_ROCM still False), so also mirror when the
|
||||
|
|
@ -2553,26 +2861,41 @@ def get_device_map(gpu_ids: Optional[list[int]] = None) -> str:
|
|||
|
||||
Returns ``"balanced"`` (shard evenly across GPUs) when:
|
||||
- ``gpu_ids`` explicitly lists >1 GPU, **or**
|
||||
- ``CUDA_VISIBLE_DEVICES`` uses UUID/MIG identifiers (non-numeric) and
|
||||
>1 GPU is visible (fallback: numeric IDs unresolvable, so assume
|
||||
multi-GPU is intended).
|
||||
- ``CUDA_VISIBLE_DEVICES``/``ZE_AFFINITY_MASK`` uses non-numeric
|
||||
identifiers (UUID/MIG/wildcard) and >1 GPU is visible (fallback:
|
||||
numeric IDs unresolvable, so assume multi-GPU is intended).
|
||||
|
||||
Returns ``"sequential"`` (single device) otherwise, including non-CUDA
|
||||
backends (CPU, MLX).
|
||||
Returns ``"sequential"`` (single device) otherwise, including CPU/MLX
|
||||
backends.
|
||||
|
||||
Use ``prepare_gpu_selection()`` upstream to determine ``gpu_ids`` -- it
|
||||
handles auto-selecting the minimum GPUs needed for a model.
|
||||
"""
|
||||
device = get_device()
|
||||
if device == DeviceType.CUDA:
|
||||
if device in (DeviceType.CUDA, DeviceType.XPU):
|
||||
multi_gpu = gpu_ids is not None and len(gpu_ids) > 1
|
||||
|
||||
if not multi_gpu:
|
||||
# UUID/MIG masks can't be split into numeric IDs; >1 visible GPU
|
||||
# means multi-GPU sharding is intended.
|
||||
parent_visible_spec = _get_parent_visible_gpu_spec()
|
||||
if parent_visible_spec["numeric_ids"] is None and get_visible_gpu_count() > 1:
|
||||
multi_gpu = True
|
||||
if device == DeviceType.CUDA:
|
||||
# UUID/MIG masks can't be split into numeric IDs; >1 visible GPU
|
||||
# means multi-GPU sharding is intended.
|
||||
if parent_visible_spec["numeric_ids"] is None and get_visible_gpu_count() > 1:
|
||||
multi_gpu = True
|
||||
elif device == DeviceType.XPU and gpu_ids is None:
|
||||
# Shard across visible XPU ordinals via HF (no mask rewrite),
|
||||
# only when no gpu_ids were passed -- an explicit gpu_ids=[0]
|
||||
# means "use exactly device 0" and must stay sequential.
|
||||
supports_physical = parent_visible_spec["supports_explicit_gpu_ids"]
|
||||
has_multiple_numeric = (
|
||||
parent_visible_spec["numeric_ids"] is not None
|
||||
and len(parent_visible_spec["numeric_ids"]) > 1
|
||||
)
|
||||
has_multiple_unresolved = (
|
||||
parent_visible_spec["numeric_ids"] is None and get_visible_gpu_count() > 1
|
||||
)
|
||||
if has_multiple_unresolved or (not supports_physical and has_multiple_numeric):
|
||||
multi_gpu = True
|
||||
|
||||
if multi_gpu:
|
||||
return "balanced"
|
||||
|
|
@ -2607,6 +2930,19 @@ def raise_if_offloaded(
|
|||
)
|
||||
|
||||
|
||||
def get_torch_device_str() -> str:
|
||||
"""
|
||||
Return the torch device string for the detected hardware.
|
||||
E.g. "cuda", "xpu", or "cpu".
|
||||
"""
|
||||
device = get_device()
|
||||
if device == DeviceType.CUDA:
|
||||
return "cuda"
|
||||
elif device == DeviceType.XPU:
|
||||
return "xpu"
|
||||
return "cpu"
|
||||
|
||||
|
||||
def safe_num_proc(desired: Optional[int] = None) -> int:
|
||||
"""
|
||||
Return a safe ``num_proc`` for ``dataset.map()`` calls.
|
||||
|
|
@ -2674,7 +3010,32 @@ def dataset_map_num_proc(desired: Optional[int] = None) -> Optional[int]:
|
|||
Returns ``None`` on spawn platforms (Windows, macOS) because ``datasets``
|
||||
treats ``num_proc=1`` as multiprocessing (creates ``Pool(1)``); only
|
||||
``num_proc=None`` guarantees in-process execution.
|
||||
|
||||
Also returns ``None`` on XPU once its runtime is initialized in this
|
||||
process: ``os.fork()`` corrupts the Level-Zero context, making Triton
|
||||
kernels fail with "Pointer argument doesn't reference XPU device memory".
|
||||
Pre-init XPU hosts can still parallelize CPU-side preprocessing.
|
||||
"""
|
||||
if sys.platform in ("win32", "darwin"):
|
||||
return None
|
||||
|
||||
if get_device() == DeviceType.XPU:
|
||||
try:
|
||||
import torch
|
||||
except Exception:
|
||||
# No torch means no active XPU runtime, so CPU-side dataset
|
||||
# parallelism is still safe.
|
||||
return safe_num_proc(desired)
|
||||
|
||||
xpu = getattr(torch, "xpu", None)
|
||||
is_initialized = getattr(xpu, "is_initialized", None)
|
||||
if callable(is_initialized):
|
||||
try:
|
||||
if is_initialized():
|
||||
return None
|
||||
except Exception as e:
|
||||
# Treat a failing probe as "runtime not touched yet" so
|
||||
# pre-init CPU preprocessing can still parallelize.
|
||||
logger.debug("torch.xpu.is_initialized() probe failed: %s", e)
|
||||
|
||||
return safe_num_proc(desired)
|
||||
|
|
|
|||
|
|
@ -294,17 +294,31 @@ def format_error_message(error: Exception, model_name: str) -> str:
|
|||
return "Invalid HF token. Please check your token and try again."
|
||||
|
||||
if (
|
||||
"memory" in error_str
|
||||
or "cuda" in error_str
|
||||
or "mlx" in error_str
|
||||
or "out of memory" in error_str
|
||||
"out of memory" in error_str
|
||||
or "out of device memory" in error_str
|
||||
or "out_of_device_memory" in error_str # ZE_RESULT_ERROR_OUT_OF_DEVICE_MEMORY
|
||||
or "out_of_host_memory" in error_str # ZE_RESULT_ERROR_OUT_OF_HOST_MEMORY
|
||||
or "not enough memory" in error_str
|
||||
or "cannot allocate memory" in error_str
|
||||
or "memory allocation failed" in error_str
|
||||
or "cublas_status_alloc_failed" in error_str # cuBLAS workspace OOM
|
||||
or ("cuda error" in error_str and "alloc" in error_str)
|
||||
or ("xpu" in error_str and ("alloc" in error_str or "memory" in error_str))
|
||||
or isinstance(error, MemoryError)
|
||||
or ("mlx" in error_str and ("memory" in error_str or "allocate" in error_str))
|
||||
):
|
||||
# Resolve get_device() at call time (not import time) so tests that
|
||||
# monkey-patch utils.hardware.get_device after this module is loaded
|
||||
# still see the patched backend.
|
||||
from utils.hardware import get_device
|
||||
|
||||
device = get_device()
|
||||
device_label = {"cuda": "GPU", "mlx": "Apple Silicon GPU", "cpu": "system"}.get(
|
||||
device.value, "GPU"
|
||||
)
|
||||
device_label = {
|
||||
"cuda": "GPU",
|
||||
"xpu": "Intel GPU",
|
||||
"mlx": "Apple Silicon GPU",
|
||||
"cpu": "system",
|
||||
}.get(device.value, "GPU")
|
||||
return f"Not enough {device_label} memory to load '{model_short}'. Try a smaller model or free memory."
|
||||
|
||||
return str(error)
|
||||
|
|
|
|||
538
tests/studio/test_xpu_spoof_pipeline.py
Normal file
538
tests/studio/test_xpu_spoof_pipeline.py
Normal file
|
|
@ -0,0 +1,538 @@
|
|||
# SPDX-License-Identifier: AGPL-3.0-only
|
||||
"""Full Intel XPU spoof pipeline: fake torch.xpu on a GPU-less/NVIDIA runner so
|
||||
Studio's hardware selection + training-device path (detect -> select -> apply ->
|
||||
device_map -> cache clear) runs exactly as the CUDA path does, with no real
|
||||
Intel hardware. The XPU sibling of tests/_zoo_aggressive_cuda_spoof.py.
|
||||
|
||||
State-sensitive: it fresh-imports the Studio hardware module under the spoof and
|
||||
mutates its module globals, so studio-backend-ci.yml runs it in the isolated
|
||||
"Hardware-spoof tests" step (never alongside tests that import hardware).
|
||||
|
||||
torch.xpu surface faked here mirrors the PyTorch 2.6+ API hardware.py calls:
|
||||
is_available, device_count, current_device, get_device_name,
|
||||
get_device_properties(idx).total_memory, memory_allocated/reserved, mem_get_info
|
||||
(incl. the Arc B580 / Lunar Lake RuntimeError), is_initialized, synchronize,
|
||||
empty_cache, plus torch.version.xpu.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
import types
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||
STUDIO_BACKEND = REPO_ROOT / "studio" / "backend"
|
||||
|
||||
|
||||
def _make_fake_xpu(
|
||||
*,
|
||||
available: bool = True,
|
||||
device_count: int = 2,
|
||||
total_gb: float = 16.0,
|
||||
used_gb: float = 1.0,
|
||||
device_name: str = "Intel(R) Arc(TM) B580 Graphics (spoofed)",
|
||||
mem_get_info: str = "ok", # "ok" | "raise" | "absent"
|
||||
is_initialized: bool = False,
|
||||
):
|
||||
"""Build a fake torch.xpu namespace + a call counter for synchronize/empty_cache.
|
||||
|
||||
mem_get_info: "ok" returns (free, total); "raise" models the Arc B580 / Lunar
|
||||
Lake "device doesn't support querying free memory" RuntimeError; "absent"
|
||||
omits the attribute so the memory_allocated fallback path is exercised.
|
||||
"""
|
||||
total_bytes = int(total_gb * 1024**3)
|
||||
used_bytes = int(used_gb * 1024**3)
|
||||
calls = {"synchronize": 0, "empty_cache": 0}
|
||||
props = types.SimpleNamespace(name = device_name, total_memory = total_bytes)
|
||||
|
||||
def _mem_get_info(idx = 0):
|
||||
if mem_get_info == "raise":
|
||||
raise RuntimeError(
|
||||
"The device (Intel(R) Arc(TM) B580 Graphics) doesn't support "
|
||||
"querying the available free memory."
|
||||
)
|
||||
return (total_bytes - used_bytes, total_bytes)
|
||||
|
||||
def _sync(*a, **k):
|
||||
calls["synchronize"] += 1
|
||||
|
||||
def _empty(*a, **k):
|
||||
calls["empty_cache"] += 1
|
||||
|
||||
xpu = types.SimpleNamespace(
|
||||
is_available = lambda: available,
|
||||
device_count = lambda: device_count,
|
||||
current_device = lambda: 0,
|
||||
get_device_name = lambda idx = 0: device_name,
|
||||
get_device_properties = lambda idx = 0: props,
|
||||
memory_allocated = lambda idx = 0: used_bytes,
|
||||
memory_reserved = lambda idx = 0: used_bytes,
|
||||
is_initialized = lambda: is_initialized,
|
||||
synchronize = _sync,
|
||||
empty_cache = _empty,
|
||||
)
|
||||
if mem_get_info != "absent":
|
||||
xpu.mem_get_info = _mem_get_info
|
||||
return xpu, calls
|
||||
|
||||
|
||||
def _import_studio_hardware_module():
|
||||
"""Fresh-import Studio's hardware module so detect_hardware re-runs under the
|
||||
current spoofs (mirrors test_hardware_dispatch_matrix.py)."""
|
||||
if str(STUDIO_BACKEND) not in sys.path:
|
||||
sys.path.insert(0, str(STUDIO_BACKEND))
|
||||
sys.modules.pop("utils.hardware.hardware", None)
|
||||
sys.modules.pop("utils.hardware", None)
|
||||
from utils.hardware import hardware as hw # type: ignore
|
||||
|
||||
return hw
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def spoof_xpu(monkeypatch):
|
||||
"""Apply a full torch.xpu spoof and return (hardware_module, xpu_call_counter).
|
||||
|
||||
Defaults present an unambiguous "prefer XPU" host: CUDA hidden, a numeric
|
||||
ZE_AFFINITY_MASK, and torch.xpu reporting devices. Override cuda_available /
|
||||
cuda_visible / force_xpu / ze_mask to model hybrid or canary hosts.
|
||||
"""
|
||||
|
||||
def _apply(
|
||||
*,
|
||||
cuda_available: bool = False,
|
||||
cuda_visible: str = "", # "" hides CUDA; None unsets; else passthrough
|
||||
ze_mask: str = "0,1", # None unsets the mask
|
||||
force_xpu: bool = False,
|
||||
xpu_version = "2.7",
|
||||
**xpu_kwargs,
|
||||
):
|
||||
import torch
|
||||
|
||||
monkeypatch.setattr(torch.cuda, "is_available", lambda: cuda_available)
|
||||
if cuda_available:
|
||||
monkeypatch.setattr(
|
||||
torch.cuda,
|
||||
"get_device_properties",
|
||||
lambda i = 0: types.SimpleNamespace(name = "Stub NVIDIA GPU"),
|
||||
raising = False,
|
||||
)
|
||||
fake_xpu, calls = _make_fake_xpu(**xpu_kwargs)
|
||||
monkeypatch.setattr(torch, "xpu", fake_xpu, raising = False)
|
||||
monkeypatch.setattr(torch.version, "xpu", xpu_version, raising = False)
|
||||
|
||||
if ze_mask is None:
|
||||
monkeypatch.delenv("ZE_AFFINITY_MASK", raising = False)
|
||||
else:
|
||||
monkeypatch.setenv("ZE_AFFINITY_MASK", ze_mask)
|
||||
if cuda_visible is None:
|
||||
monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising = False)
|
||||
else:
|
||||
monkeypatch.setenv("CUDA_VISIBLE_DEVICES", cuda_visible)
|
||||
if force_xpu:
|
||||
monkeypatch.setenv("UNSLOTH_FORCE_XPU", "1")
|
||||
else:
|
||||
monkeypatch.delenv("UNSLOTH_FORCE_XPU", raising = False)
|
||||
# FLAT is the oneAPI default; pin it so the test is host-independent.
|
||||
monkeypatch.delenv("ZE_FLAT_DEVICE_HIERARCHY", raising = False)
|
||||
|
||||
hw = _import_studio_hardware_module()
|
||||
hw._visible_gpu_count = None
|
||||
return hw, calls
|
||||
|
||||
return _apply
|
||||
|
||||
|
||||
# ---------- detection ----------
|
||||
|
||||
|
||||
def test_detect_hardware_routes_to_xpu(spoof_xpu):
|
||||
hw, _ = spoof_xpu()
|
||||
assert hw.detect_hardware() == hw.DeviceType.XPU
|
||||
assert hw.CHAT_ONLY is False
|
||||
assert hw.IS_ROCM is False
|
||||
|
||||
|
||||
def test_force_xpu_env_routes_to_xpu_even_without_mask(spoof_xpu):
|
||||
hw, _ = spoof_xpu(force_xpu = True, ze_mask = None, cuda_visible = None)
|
||||
assert hw.detect_hardware() == hw.DeviceType.XPU
|
||||
|
||||
|
||||
def test_bare_mask_with_cuda_present_stays_cuda(spoof_xpu):
|
||||
# Canary: a stray inherited ZE_AFFINITY_MASK must NOT steal a CUDA host.
|
||||
hw, _ = spoof_xpu(cuda_available = True, cuda_visible = None, ze_mask = "0,1")
|
||||
assert hw.detect_hardware() == hw.DeviceType.CUDA
|
||||
|
||||
|
||||
def test_force_xpu_on_hybrid_hides_cuda_for_workers(spoof_xpu):
|
||||
# Forced XPU with CUDA still visible must hide CUDA: unsloth's
|
||||
# device_type picks CUDA before XPU and ignores UNSLOTH_FORCE_XPU,
|
||||
# so workers would otherwise silently train on CUDA.
|
||||
hw, _ = spoof_xpu(force_xpu = True, cuda_available = True, cuda_visible = None, ze_mask = None)
|
||||
assert hw.detect_hardware() == hw.DeviceType.XPU
|
||||
import os
|
||||
|
||||
assert os.environ["CUDA_VISIBLE_DEVICES"] == ""
|
||||
|
||||
|
||||
def test_force_xpu_without_working_xpu_leaves_cuda_untouched(spoof_xpu):
|
||||
# Canary: FORCE_XPU on a CUDA host with no working XPU must fall
|
||||
# through to CUDA and must NOT hide it.
|
||||
hw, _ = spoof_xpu(
|
||||
force_xpu = True,
|
||||
cuda_available = True,
|
||||
cuda_visible = None,
|
||||
ze_mask = None,
|
||||
available = False,
|
||||
)
|
||||
assert hw.detect_hardware() == hw.DeviceType.CUDA
|
||||
import os
|
||||
|
||||
assert "CUDA_VISIBLE_DEVICES" not in os.environ
|
||||
|
||||
|
||||
def test_apply_gpu_ids_predetect_never_probes_torch(spoof_xpu, monkeypatch):
|
||||
# Workers call apply_gpu_ids() BEFORE detect_hardware(); a lazy detect
|
||||
# would probe torch.cuda against the unmasked parent env, latching device
|
||||
# enumeration before the mask is written. Pre-detect it must decide from
|
||||
# env/build attributes only.
|
||||
import torch
|
||||
|
||||
hw, _ = spoof_xpu(ze_mask = None, cuda_visible = None)
|
||||
assert hw.DEVICE is None # fresh import, pre-detect
|
||||
|
||||
def _poisoned_detect():
|
||||
raise AssertionError("apply_gpu_ids triggered detect_hardware pre-mask")
|
||||
|
||||
monkeypatch.setattr(hw, "detect_hardware", _poisoned_detect)
|
||||
monkeypatch.setattr(torch.cuda, "is_available", _poisoned_detect, raising = False)
|
||||
# CUDA-build torch (torch.version.cuda set on this box or spoofed):
|
||||
monkeypatch.setattr(torch.version, "cuda", "12.8", raising = False)
|
||||
monkeypatch.setattr(torch.version, "xpu", None, raising = False)
|
||||
hw.apply_gpu_ids([1])
|
||||
import os
|
||||
|
||||
assert os.environ["CUDA_VISIBLE_DEVICES"] == "1"
|
||||
assert "ZE_AFFINITY_MASK" not in os.environ
|
||||
|
||||
|
||||
def test_apply_gpu_ids_predetect_xpu_build_writes_ze_mask(spoof_xpu, monkeypatch):
|
||||
# Pre-detect on an XPU-build torch (version.xpu set, no cuda/hip):
|
||||
# the mask must go to ZE_AFFINITY_MASK without any runtime probe.
|
||||
import torch
|
||||
|
||||
hw, _ = spoof_xpu(ze_mask = None, cuda_visible = None)
|
||||
assert hw.DEVICE is None
|
||||
|
||||
def _poisoned_detect():
|
||||
raise AssertionError("apply_gpu_ids triggered detect_hardware pre-mask")
|
||||
|
||||
monkeypatch.setattr(hw, "detect_hardware", _poisoned_detect)
|
||||
monkeypatch.setattr(torch.version, "cuda", None, raising = False)
|
||||
monkeypatch.setattr(torch.version, "hip", None, raising = False)
|
||||
monkeypatch.setattr(torch.version, "xpu", "2.7", raising = False)
|
||||
hw.apply_gpu_ids([0])
|
||||
import os
|
||||
|
||||
assert os.environ["ZE_AFFINITY_MASK"] == "0"
|
||||
assert "CUDA_VISIBLE_DEVICES" not in os.environ
|
||||
|
||||
|
||||
def test_apply_gpu_ids_predetect_xpu_compiled_with_null_version(spoof_xpu, monkeypatch):
|
||||
# version.xpu can be None on a working XPU build; torch.xpu._is_compiled()
|
||||
# must be accepted as the build signal so the mask still goes to
|
||||
# ZE_AFFINITY_MASK.
|
||||
import torch
|
||||
|
||||
hw, _ = spoof_xpu(ze_mask = None, cuda_visible = None)
|
||||
assert hw.DEVICE is None
|
||||
monkeypatch.setattr(
|
||||
hw, "detect_hardware", lambda: (_ for _ in ()).throw(AssertionError("detect ran"))
|
||||
)
|
||||
monkeypatch.setattr(torch.version, "cuda", None, raising = False)
|
||||
monkeypatch.setattr(torch.version, "hip", None, raising = False)
|
||||
monkeypatch.setattr(torch.version, "xpu", None, raising = False)
|
||||
monkeypatch.setattr(torch.xpu, "_is_compiled", lambda: True, raising = False)
|
||||
hw.apply_gpu_ids([0])
|
||||
import os
|
||||
|
||||
assert os.environ["ZE_AFFINITY_MASK"] == "0"
|
||||
assert "CUDA_VISIBLE_DEVICES" not in os.environ
|
||||
|
||||
|
||||
def test_apply_gpu_ids_predetect_force_on_cuda_build_writes_cvd(spoof_xpu, monkeypatch):
|
||||
# UNSLOTH_FORCE_XPU=1 on a CUDA build (no XPU compiled in): detect falls
|
||||
# back to CUDA, so the pre-detect mask must go to CUDA_VISIBLE_DEVICES,
|
||||
# not ZE_AFFINITY_MASK.
|
||||
import torch
|
||||
|
||||
hw, _ = spoof_xpu(force_xpu = True, ze_mask = None, cuda_visible = None)
|
||||
assert hw.DEVICE is None
|
||||
monkeypatch.setattr(
|
||||
hw, "detect_hardware", lambda: (_ for _ in ()).throw(AssertionError("detect ran"))
|
||||
)
|
||||
monkeypatch.setattr(torch.version, "cuda", "12.8", raising = False)
|
||||
monkeypatch.setattr(torch.version, "xpu", None, raising = False)
|
||||
monkeypatch.setattr(torch.xpu, "_is_compiled", lambda: False, raising = False)
|
||||
hw.apply_gpu_ids([1])
|
||||
import os
|
||||
|
||||
assert os.environ["CUDA_VISIBLE_DEVICES"] == "1"
|
||||
assert "ZE_AFFINITY_MASK" not in os.environ
|
||||
|
||||
|
||||
def test_apply_gpu_ids_predetect_dual_build_honors_xpu_hint(spoof_xpu, monkeypatch):
|
||||
# Dual CUDA+XPU build launched the documented XPU way (CUDA hidden + ZE
|
||||
# mask): the mask must narrow ZE_AFFINITY_MASK, not re-expose the hidden
|
||||
# CUDA via CUDA_VISIBLE_DEVICES. Mirrors detect_hardware's hint.
|
||||
import torch
|
||||
|
||||
hw, _ = spoof_xpu(ze_mask = "0,1", cuda_visible = "")
|
||||
assert hw.DEVICE is None
|
||||
monkeypatch.setattr(
|
||||
hw, "detect_hardware", lambda: (_ for _ in ()).throw(AssertionError("detect ran"))
|
||||
)
|
||||
monkeypatch.setattr(torch.version, "cuda", "12.8", raising = False)
|
||||
monkeypatch.setattr(torch.version, "xpu", "2.7", raising = False)
|
||||
hw.apply_gpu_ids([0])
|
||||
import os
|
||||
|
||||
assert os.environ["ZE_AFFINITY_MASK"] == "0"
|
||||
assert os.environ["CUDA_VISIBLE_DEVICES"] == "" # stays hidden
|
||||
|
||||
|
||||
def test_apply_gpu_ids_predetect_dual_build_cuda_active_writes_cvd(spoof_xpu, monkeypatch):
|
||||
# Canary: dual build with CUDA active (no hint) keeps CUDA masking, same
|
||||
# as detect_hardware picking CUDA on a hybrid host.
|
||||
import torch
|
||||
|
||||
hw, _ = spoof_xpu(ze_mask = "0,1", cuda_visible = None)
|
||||
assert hw.DEVICE is None
|
||||
monkeypatch.setattr(
|
||||
hw, "detect_hardware", lambda: (_ for _ in ()).throw(AssertionError("detect ran"))
|
||||
)
|
||||
monkeypatch.setattr(torch.version, "cuda", "12.8", raising = False)
|
||||
monkeypatch.setattr(torch.version, "xpu", "2.7", raising = False)
|
||||
hw.apply_gpu_ids([1])
|
||||
import os
|
||||
|
||||
assert os.environ["CUDA_VISIBLE_DEVICES"] == "1"
|
||||
assert os.environ["ZE_AFFINITY_MASK"] == "0,1" # untouched
|
||||
|
||||
|
||||
def test_apply_gpu_ids_trusts_parent_backend_param(spoof_xpu, monkeypatch):
|
||||
# Workers pass the parent's detected backend (config["device_backend"]):
|
||||
# it must win over build heuristics in both directions, mirroring
|
||||
# detect_hardware's availability check and CUDA fallback exactly.
|
||||
import torch
|
||||
|
||||
hw, _ = spoof_xpu(ze_mask = None, cuda_visible = None, force_xpu = True)
|
||||
assert hw.DEVICE is None
|
||||
monkeypatch.setattr(
|
||||
hw, "detect_hardware", lambda: (_ for _ in ()).throw(AssertionError("detect ran"))
|
||||
)
|
||||
# Forced XPU + XPU build, but the parent detected CUDA (xpu had no
|
||||
# device): backend="cuda" must route to CUDA_VISIBLE_DEVICES.
|
||||
monkeypatch.setattr(torch.version, "cuda", "12.8", raising = False)
|
||||
monkeypatch.setattr(torch.version, "xpu", "2.7", raising = False)
|
||||
hw.apply_gpu_ids([1], backend = "cuda")
|
||||
import os
|
||||
|
||||
assert os.environ["CUDA_VISIBLE_DEVICES"] == "1"
|
||||
assert "ZE_AFFINITY_MASK" not in os.environ
|
||||
|
||||
# And backend="xpu" routes to ZE_AFFINITY_MASK even on a CUDA build.
|
||||
monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising = False)
|
||||
monkeypatch.setattr(torch.version, "xpu", None, raising = False)
|
||||
hw.apply_gpu_ids([0], backend = "xpu")
|
||||
assert os.environ["ZE_AFFINITY_MASK"] == "0"
|
||||
assert "CUDA_VISIBLE_DEVICES" not in os.environ
|
||||
|
||||
|
||||
def test_apply_gpu_ids_predetect_hidden_cuda_without_mask_prefers_xpu(spoof_xpu, monkeypatch):
|
||||
# Hidden CUDA on an XPU-capable build prefers XPU even with NO ZE mask
|
||||
# set (detection falls through to XPU in that state); writing the ids to
|
||||
# CUDA_VISIBLE_DEVICES would re-expose the hidden CUDA.
|
||||
import torch
|
||||
|
||||
hw, _ = spoof_xpu(ze_mask = None, cuda_visible = "")
|
||||
assert hw.DEVICE is None
|
||||
monkeypatch.setattr(
|
||||
hw, "detect_hardware", lambda: (_ for _ in ()).throw(AssertionError("detect ran"))
|
||||
)
|
||||
monkeypatch.setattr(torch.version, "cuda", "12.8", raising = False)
|
||||
monkeypatch.setattr(torch.version, "xpu", "2.7", raising = False)
|
||||
hw.apply_gpu_ids([0])
|
||||
import os
|
||||
|
||||
assert os.environ["ZE_AFFINITY_MASK"] == "0"
|
||||
assert os.environ["CUDA_VISIBLE_DEVICES"] == "" # stays hidden
|
||||
|
||||
|
||||
# ---------- visibility / selection ----------
|
||||
|
||||
|
||||
def test_apply_gpu_ids_writes_ze_affinity_mask(spoof_xpu, monkeypatch):
|
||||
hw, _ = spoof_xpu()
|
||||
hw.detect_hardware()
|
||||
monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "sentinel")
|
||||
hw.apply_gpu_ids([0, 1])
|
||||
import os
|
||||
|
||||
assert os.environ["ZE_AFFINITY_MASK"] == "0,1"
|
||||
# XPU pinning must not touch CUDA_VISIBLE_DEVICES (hybrid-host safety).
|
||||
assert os.environ["CUDA_VISIBLE_DEVICES"] == "sentinel"
|
||||
|
||||
|
||||
def test_get_visible_gpu_count_uses_device_count(spoof_xpu):
|
||||
hw, _ = spoof_xpu(ze_mask = "0,1", device_count = 2)
|
||||
hw.detect_hardware()
|
||||
assert hw.get_visible_gpu_count() == 2
|
||||
|
||||
|
||||
def test_get_visible_gpu_count_empty_mask_is_zero(spoof_xpu):
|
||||
hw, _ = spoof_xpu(ze_mask = "")
|
||||
hw.detect_hardware()
|
||||
assert hw.get_visible_gpu_count() == 0
|
||||
|
||||
|
||||
def test_flat_numeric_mask_reports_relative_ordinals(spoof_xpu):
|
||||
hw, _ = spoof_xpu(ze_mask = "4,7", device_count = 2)
|
||||
hw.detect_hardware()
|
||||
|
||||
spec = hw._get_parent_visible_gpu_spec()
|
||||
assert spec["numeric_ids"] is None
|
||||
assert spec["supports_explicit_gpu_ids"] is False
|
||||
|
||||
for result in (hw.get_visible_gpu_utilization(), hw.get_backend_visible_gpu_info()):
|
||||
assert result["available"] is True
|
||||
assert result["index_kind"] == "relative"
|
||||
assert result["parent_visible_gpu_ids"] == []
|
||||
assert [device["index"] for device in result["devices"]] == [0, 1]
|
||||
|
||||
|
||||
def test_composite_numeric_mask_reports_physical_ids(spoof_xpu, monkeypatch):
|
||||
hw, _ = spoof_xpu(ze_mask = "4,7", device_count = 2)
|
||||
monkeypatch.setenv("ZE_FLAT_DEVICE_HIERARCHY", "COMPOSITE")
|
||||
hw.detect_hardware()
|
||||
|
||||
spec = hw._get_parent_visible_gpu_spec()
|
||||
assert spec["numeric_ids"] == [4, 7]
|
||||
assert spec["supports_explicit_gpu_ids"] is True
|
||||
|
||||
for result in (hw.get_visible_gpu_utilization(), hw.get_backend_visible_gpu_info()):
|
||||
assert result["available"] is True
|
||||
assert result["index_kind"] == "physical"
|
||||
assert result["parent_visible_gpu_ids"] == [4, 7]
|
||||
assert [device["index"] for device in result["devices"]] == [4, 7]
|
||||
|
||||
|
||||
def test_get_device_map_multi_is_balanced(spoof_xpu):
|
||||
hw, _ = spoof_xpu(ze_mask = "0,1", device_count = 2)
|
||||
hw.detect_hardware()
|
||||
assert hw.get_device_map([0, 1]) == "balanced"
|
||||
|
||||
|
||||
def test_get_device_map_explicit_single_is_sequential(spoof_xpu):
|
||||
hw, _ = spoof_xpu(ze_mask = "0,1", device_count = 2)
|
||||
hw.detect_hardware()
|
||||
# Explicit gpu_ids=[0] is a deliberate single-device request.
|
||||
assert hw.get_device_map([0]) == "sequential"
|
||||
|
||||
|
||||
# ---------- cache / telemetry / versions ----------
|
||||
|
||||
|
||||
def test_clear_gpu_cache_calls_xpu(spoof_xpu):
|
||||
hw, calls = spoof_xpu()
|
||||
hw.detect_hardware()
|
||||
hw.clear_gpu_cache()
|
||||
assert calls["synchronize"] >= 1
|
||||
assert calls["empty_cache"] >= 1
|
||||
|
||||
|
||||
def test_package_versions_survive_broken_xpu_runtime(spoof_xpu, monkeypatch):
|
||||
# A broken Intel runtime raising in is_available() must not blank the
|
||||
# CUDA/ROCm versions on NVIDIA/AMD hosts.
|
||||
import torch
|
||||
|
||||
hw, _ = spoof_xpu(cuda_available = True, cuda_visible = None, ze_mask = None)
|
||||
monkeypatch.setattr(torch.version, "cuda", "12.8", raising = False)
|
||||
|
||||
def _broken():
|
||||
raise RuntimeError("Level Zero init failed")
|
||||
|
||||
monkeypatch.setattr(torch.xpu, "is_available", _broken)
|
||||
versions = hw.get_package_versions()
|
||||
assert versions["cuda"] == "12.8"
|
||||
assert versions.get("xpu") is None
|
||||
|
||||
|
||||
def test_package_versions_reports_xpu(spoof_xpu):
|
||||
hw, _ = spoof_xpu(xpu_version = "2.7")
|
||||
hw.detect_hardware()
|
||||
assert hw.get_package_versions().get("xpu") == "2.7"
|
||||
|
||||
|
||||
def test_package_versions_xpu_available_fallback(spoof_xpu):
|
||||
hw, _ = spoof_xpu(xpu_version = None)
|
||||
hw.detect_hardware()
|
||||
assert hw.get_package_versions().get("xpu") == "available"
|
||||
|
||||
|
||||
def test_per_device_info_mem_get_info_ok(spoof_xpu):
|
||||
hw, _ = spoof_xpu(total_gb = 16.0, used_gb = 1.0)
|
||||
hw.detect_hardware()
|
||||
info = hw._torch_get_per_device_info([0])
|
||||
assert len(info) == 1
|
||||
assert info[0]["total_gb"] == pytest.approx(16.0, abs = 0.1)
|
||||
assert info[0]["used_gb"] == pytest.approx(1.0, abs = 0.1)
|
||||
|
||||
|
||||
def test_mem_get_info_runtimeerror_keeps_device_with_unknown_usage(spoof_xpu):
|
||||
# Arc B580 and Lunar Lake can reject mem_get_info while remaining usable.
|
||||
hw, _ = spoof_xpu(mem_get_info = "raise", total_gb = 16.0, device_count = 2)
|
||||
hw.detect_hardware()
|
||||
|
||||
info = hw._torch_get_per_device_info([0])
|
||||
assert len(info) == 1
|
||||
assert info[0]["total_gb"] == pytest.approx(16.0, abs = 0.1)
|
||||
assert info[0]["used_gb"] is None
|
||||
|
||||
utilization = hw.get_visible_gpu_utilization()
|
||||
assert utilization["available"] is True
|
||||
assert len(utilization["devices"]) == 2
|
||||
assert all(device["vram_total_gb"] == pytest.approx(16.0) for device in utilization["devices"])
|
||||
assert all(device["vram_used_gb"] is None for device in utilization["devices"])
|
||||
|
||||
visibility = hw.get_backend_visible_gpu_info()
|
||||
assert visibility["available"] is True
|
||||
assert len(visibility["devices"]) == 2
|
||||
assert all(device["memory_total_gb"] == pytest.approx(16.0) for device in visibility["devices"])
|
||||
|
||||
|
||||
def test_per_device_info_no_mem_get_info_uses_none(spoof_xpu):
|
||||
hw, _ = spoof_xpu(mem_get_info = "absent")
|
||||
hw.detect_hardware()
|
||||
info = hw._torch_get_per_device_info([0])
|
||||
assert len(info) == 1
|
||||
assert info[0]["used_gb"] is None
|
||||
|
||||
|
||||
# ---------- training-device wiring ----------
|
||||
|
||||
|
||||
def test_get_torch_device_str_is_xpu(spoof_xpu):
|
||||
hw, _ = spoof_xpu()
|
||||
hw.detect_hardware()
|
||||
assert hw.get_torch_device_str() == "xpu"
|
||||
|
||||
|
||||
def test_dataset_map_num_proc_none_after_xpu_init(spoof_xpu):
|
||||
# os.fork() after Level-Zero init corrupts the XPU context -> force in-process.
|
||||
hw, _ = spoof_xpu(is_initialized = True)
|
||||
hw.detect_hardware()
|
||||
assert hw.dataset_map_num_proc(4) is None
|
||||
Loading…
Add table
Add a link
Reference in a new issue