unsloth/studio/backend/tests/test_torchao_select.py
Ayushman c356427f30
Guard Windows ROCm torchao override skip (#6837)
* Fix: skip fp16/bf16 validation for full finetuning in RL trainers

When doing full finetuning (FFT) of a bfloat16 model, the fp16/bf16
mismatch validation fires before the corrective logic runs, causing a
misleading error even though the code would properly handle it downstream.
Skip the validation when full_finetuning is active.

Fixes #6731

* Fix: auto-correct fp16/bf16 mismatches for full finetuning before validation

Instead of entirely skipping validation (which could let mismatches
through when mixed_precision_dtype is float32), auto-correct explicit
fp16/bf16 settings that conflict with the model's dtype for FFT. This
way the existing validation still catches real mismatches for non-FFT
cases, and the corrective logic below handles the normalized settings.

Fixes the issue raised in Codex review of PR #6813.

* Guard Windows ROCm torchao override skip

Detect installed ROCm torch directly before applying the torchao override so Windows ROCm environments never install the crashing torchao package even if the earlier ROCm-installed flag is missing.

* Update unsloth/models/rl.py

Co-authored-by: gemini-code-assist[bot] <176961590+gemini-code-assist[bot]@users.noreply.github.com>

* Update studio/install_python_stack.py

Co-authored-by: gemini-code-assist[bot] <176961590+gemini-code-assist[bot]@users.noreply.github.com>

* Harden ROCm probe and sync RL precision flags

Tolerate stray stdout noise when probing Windows ROCm torch installs by checking the last non-empty output line, matching the existing torch version probe behavior. Also keep args.fp16 and args.bf16 synchronized with the full-finetuning precision auto-corrections in the RL trainer patch so downstream eval settings see a consistent TrainingArguments state.

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* Add MLX trainer compatibility shims

Patch imported MLXTrainer and MLXTrainingConfig objects to preserve the expected dataclass field ordering and to provide a _train_dataset_for_batches fallback when older trainers or test doubles only expose train_dataset. Also add focused worker tests covering both compatibility paths.

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* Scope PR to Windows ROCm torchao guard

* Restore PR scope to Windows ROCm guard

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* test: cover Windows ROCm torchao skip behavior

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

---------

Co-authored-by: Ayushman Paul <ayushman@HP>
Co-authored-by: gemini-code-assist[bot] <176961590+gemini-code-assist[bot]@users.noreply.github.com>
Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
Co-authored-by: Lee Jackson <130007945+Imagineer99@users.noreply.github.com>
Co-authored-by: imagineer99 <samleejackson0@gmail.com>
2026-07-03 19:24:29 +01:00

129 lines
5.2 KiB
Python

# SPDX-License-Identifier: AGPL-3.0-only
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
"""Tests for _select_torchao_spec in install_python_stack.py.
torchao's C++ extensions are built against one exact torch release, so the
installer must pick the torchao version matching the torch installed in the
venv (otherwise the cpp kernels are skipped). This pins that mapping.
"""
from __future__ import annotations
import sys
from pathlib import Path
from unittest.mock import MagicMock
import pytest
# install_python_stack.py lives at repo_root/studio/install_python_stack.py
_INSTALL_SCRIPT = Path(__file__).resolve().parents[2] / "install_python_stack.py"
def _load_module(monkeypatch):
"""(Re-)import install_python_stack and return it (mirrors test_pytorch_mirror)."""
sys.modules.pop("install_python_stack", None)
monkeypatch.syspath_prepend(str(_INSTALL_SCRIPT.parent))
import install_python_stack
return install_python_stack
@pytest.mark.parametrize(
"torch_version, expected",
[
# torch 2.10 (the reported bug: cu130 resolves 2.10.0) -> 0.16.0,
# independent of the local +cuXXX/+rocm/+cpu suffix or patch level.
("2.10.0+cu130", "torchao==0.16.0"),
("2.10.0+rocm6.4", "torchao==0.16.0"),
("2.10.0+cpu", "torchao==0.16.0"),
("2.10.1", "torchao==0.16.0"),
("2.10.0", "torchao==0.16.0"),
# Pre-release / dev / rc builds: the minor is cleaned of non-digits.
("2.10.0rc1", "torchao==0.16.0"),
("2.10.0.dev20250804+cu130", "torchao==0.16.0"),
("2.10rc1", "torchao==0.16.0"),
# torch 2.11 (reachable via ROCm rocm7.2) and forward -> 0.17.0.
("2.11.0+cu130", "torchao==0.17.0"),
("2.11.0", "torchao==0.17.0"),
("2.12.0", "torchao==0.17.0"),
# torch <=2.9 keeps today's pin (already a correct match for 2.9.0).
("2.9.0+cu128", "torchao==0.14.0"),
("2.9.1", "torchao==0.14.0"),
("2.8.0", "torchao==0.14.0"),
("2.4.0", "torchao==0.14.0"),
# Unparseable / missing / non-2.x major -> conservative default.
(None, "torchao==0.14.0"),
("", "torchao==0.14.0"),
("garbage", "torchao==0.14.0"),
("2", "torchao==0.14.0"),
("3.0.0", "torchao==0.14.0"),
],
)
def test_select_torchao_spec(monkeypatch, torch_version, expected):
mod = _load_module(monkeypatch)
assert mod._select_torchao_spec(torch_version) == expected
def test_default_spec_matches_table(monkeypatch):
"""The default/floor stays the historical pin so older torch is unchanged."""
mod = _load_module(monkeypatch)
assert mod._TORCHAO_DEFAULT_SPEC == "torchao==0.14.0"
assert mod._select_torchao_spec("2.9.0") == mod._TORCHAO_DEFAULT_SPEC
@pytest.mark.parametrize(
("rocm_windows_torch_installed", "installed_torch_is_windows_rocm"),
[
(True, False),
(False, True),
],
)
def test_skips_torchao_on_windows_rocm(
monkeypatch, tmp_path, rocm_windows_torch_installed, installed_torch_is_windows_rocm
):
"""The overrides step must skip torchao on Windows ROCm: no working build exists
there (it imports an absent c10d backend and crashes transformers.quantizers),
so the installer skips it and relies on the runtime stub instead."""
mod = _load_module(monkeypatch)
installed_specs: list[str] = []
progress_labels: list[str] = []
def _record_pip_install(*args, **kwargs):
installed_specs.extend(str(arg) for arg in args)
return 0
unstructured_plugin = tmp_path / "unstructured"
github_plugin = tmp_path / "github"
unstructured_plugin.mkdir()
github_plugin.mkdir()
subprocess_result = MagicMock()
subprocess_result.returncode = 0
subprocess_result.stdout = ""
monkeypatch.setenv("SKIP_STUDIO_BASE", "1")
monkeypatch.setattr(mod, "IS_WINDOWS", True)
monkeypatch.setattr(mod, "IS_MACOS", False)
monkeypatch.setattr(mod, "IS_MAC_ARM", False)
monkeypatch.setattr(mod, "NO_TORCH", False)
monkeypatch.setattr(mod, "_rocm_windows_torch_installed", rocm_windows_torch_installed)
monkeypatch.setattr(
mod, "_installed_torch_is_windows_rocm", lambda: installed_torch_is_windows_rocm
)
monkeypatch.setattr(mod, "_bootstrap_uv", lambda: False)
monkeypatch.setattr(mod, "_repair_bad_anyio", lambda: None)
monkeypatch.setattr(mod, "_ensure_rocm_torch", lambda: None)
monkeypatch.setattr(mod, "_ensure_cuda_torch", lambda: None)
monkeypatch.setattr(mod, "_has_usable_nvidia_gpu", lambda: True)
monkeypatch.setattr(mod, "run", lambda *args, **kwargs: None)
monkeypatch.setattr(mod, "pip_install", _record_pip_install)
monkeypatch.setattr(mod, "_progress", lambda label: progress_labels.append(label))
monkeypatch.setattr(mod, "LOCAL_DD_UNSTRUCTURED_PLUGIN", unstructured_plugin)
monkeypatch.setattr(mod, "LOCAL_DD_GITHUB_PLUGIN", github_plugin)
monkeypatch.setattr(mod.subprocess, "run", lambda *args, **kwargs: subprocess_result)
assert mod.install_python_stack() == 0
assert not any(spec.startswith("torchao") for spec in installed_specs)
assert "dependency overrides (skipped, Windows ROCm)" in progress_labels