unsloth/studio/backend/tests/test_diffusion_engine_router.py
Daniel Han a9e5a80654 Address the round of Codex review findings on the merged diffusion phases
Memory planning and dense-quant path: size a local diffusers base's
resident companions from its on-disk VAE and text-encoder weights instead
of folding them to zero, feed the distilled variant hint into the runtime
headroom estimate so turbo and schnell models are not over-reserved, place
group-offload companions resident before attaching the transformer hooks
so a failed placement falls back to whole-module offload instead of
crashing, and bail out of the dense transformer download before it starts
when the requested quant scheme is unsupported so the load falls back to
GGUF cleanly.

sd.cpp stack: scrub the native path lease secret from sd-cli child env,
redact native load-progress errors, forward the resolved accelerator when
auto-installing a forced-native binary, release stale diffusion GPU
ownership on CPU-native loads, and remove the sd.cpp install tree on
uninstall.

Prequant and scripts: reject prequant artifacts missing base_model_id
when a base is requested, expanduser before checkpoint existence checks,
record and validate the int8 exclusion filter and fp8 fast-accum in
checkpoint metadata, make verify_prequant_backend allowlist its local
checkpoint and fail on missing or bad LPIPS and on load-peak regressions,
average only finite PSNR values in diffusion_quality, and reset the
process-wide attention backend between perf probe variants.

API and UI: normalize attention_backend casing before Literal validation,
close hidden popovers when leaving the Images page, and clear the stale
quant label when loading a direct local GGUF file.
2026-07-02 03:29:18 +00:00

182 lines
6.8 KiB
Python

# SPDX-License-Identifier: AGPL-3.0-only
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
"""Tests for the diffusion engine router (diffusers vs native sd.cpp selection)."""
from __future__ import annotations
from types import SimpleNamespace
import pytest
from core.inference import diffusion_engine_router as r
from core.inference.diffusion_families import detect_family
from core.inference.sd_cpp_engine import ENGINE_DIFFUSERS, ENGINE_SD_CPP
_ENVS = (
"UNSLOTH_DIFFUSION_ENGINE",
"UNSLOTH_DIFFUSION_SD_CPP",
"UNSLOTH_DIFFUSION_SD_CPP_MPS",
"UNSLOTH_DIFFUSION_SD_CPP_INSTALL",
)
@pytest.fixture(autouse = True)
def _clean_env_and_state(monkeypatch):
for e in _ENVS:
monkeypatch.delenv(e, raising = False)
# A light status-capable stub so neither selection nor active_status() imports the
# heavy diffusers/sd.cpp backends; the active engine NAME comes from module state.
monkeypatch.setattr(
r,
"get_active_diffusion_engine",
lambda: SimpleNamespace(status = lambda: {"loaded": False, "repo_id": None}),
)
yield
def _set_device(monkeypatch, backend):
monkeypatch.setattr(
r,
"resolve_diffusion_device_target",
lambda: SimpleNamespace(backend = backend, device = backend),
)
def _set_binary(monkeypatch, path):
monkeypatch.setattr(r, "ensure_sd_cpp_binary", lambda **_: path)
def _set_runnable(monkeypatch, version = "sd-cli v0"):
"""Stub the runnability probe so a stubbed binary path is treated as executable
(the router now probes ``SdCppEngine(...).version()`` before committing to native)."""
monkeypatch.setattr(r, "SdCppEngine", lambda **_: SimpleNamespace(version = lambda: version))
def _select(fam_name = "z-image"):
"""Activate the engine for a family and return which engine was chosen."""
r.select_and_activate_engine(detect_family(fam_name))
return r.active_engine_name()
# ── core selection matrix ─────────────────────────────────────────────────────
def test_cpu_with_binary_and_supported_family_picks_sd_cpp(monkeypatch):
_set_device(monkeypatch, "cpu")
_set_binary(monkeypatch, "/usr/bin/sd-cli")
_set_runnable(monkeypatch)
assert _select() == ENGINE_SD_CPP
assert r.active_engine_name() == ENGINE_SD_CPP
def test_present_but_not_runnable_binary_falls_back(monkeypatch):
# A binary that exists but cannot run (version() -> None) must fall back to
# diffusers at selection, not commit native and fail inside the load.
_set_device(monkeypatch, "cpu")
_set_binary(monkeypatch, "/usr/bin/sd-cli")
monkeypatch.setattr(r, "SdCppEngine", lambda **_: SimpleNamespace(version = lambda: None))
assert _select() == ENGINE_DIFFUSERS
assert "binary unavailable" in (r.active_status()["fallback_reason"] or "")
@pytest.mark.parametrize("gpu", ["cuda", "rocm", "xpu"])
def test_gpu_backends_use_diffusers(monkeypatch, gpu):
_set_device(monkeypatch, gpu)
_set_binary(monkeypatch, "/usr/bin/sd-cli") # even with a binary, GPU stays diffusers
assert _select() == ENGINE_DIFFUSERS
assert "uses diffusers" in (r.active_status()["fallback_reason"] or "")
def test_forced_diffusers_overrides_cpu(monkeypatch):
_set_device(monkeypatch, "cpu")
_set_binary(monkeypatch, "/usr/bin/sd-cli")
monkeypatch.setenv("UNSLOTH_DIFFUSION_ENGINE", "diffusers")
assert _select() == ENGINE_DIFFUSERS
assert "forced" in (r.active_status()["fallback_reason"] or "")
def test_sd_cpp_disabled_uses_diffusers(monkeypatch):
_set_device(monkeypatch, "cpu")
_set_binary(monkeypatch, "/usr/bin/sd-cli")
monkeypatch.setenv("UNSLOTH_DIFFUSION_SD_CPP", "0")
assert _select() == ENGINE_DIFFUSERS
assert "disabled" in (r.active_status()["fallback_reason"] or "")
def test_mps_default_diffusers_but_optin_sd_cpp(monkeypatch):
_set_device(monkeypatch, "mps")
_set_binary(monkeypatch, "/usr/bin/sd-cli")
_set_runnable(monkeypatch)
# Default: MPS is not native-eligible -> diffusers.
assert _select() == ENGINE_DIFFUSERS
# Opt in: MPS routes to sd.cpp.
monkeypatch.setenv("UNSLOTH_DIFFUSION_SD_CPP_MPS", "1")
assert _select() == ENGINE_SD_CPP
def test_unsupported_family_falls_back(monkeypatch):
_set_device(monkeypatch, "cpu")
_set_binary(monkeypatch, "/usr/bin/sd-cli")
monkeypatch.setattr(r, "family_sd_cpp_supported", lambda fam: False)
assert _select() == ENGINE_DIFFUSERS
assert "no native sd.cpp asset mapping" in (r.active_status()["fallback_reason"] or "")
def test_missing_binary_falls_back(monkeypatch):
_set_device(monkeypatch, "cpu")
_set_binary(monkeypatch, None) # install unavailable
assert _select() == ENGINE_DIFFUSERS
assert "binary unavailable" in (r.active_status()["fallback_reason"] or "")
def test_force_sd_cpp_on_gpu_when_binary_present(monkeypatch):
_set_device(monkeypatch, "cuda")
_set_binary(monkeypatch, "/usr/bin/sd-cli")
_set_runnable(monkeypatch)
monkeypatch.setenv("UNSLOTH_DIFFUSION_ENGINE", "sd_cpp")
assert _select() == ENGINE_SD_CPP
def test_force_sd_cpp_without_binary_falls_back(monkeypatch):
_set_device(monkeypatch, "cuda")
_set_binary(monkeypatch, None)
monkeypatch.setenv("UNSLOTH_DIFFUSION_ENGINE", "sd_cpp")
assert _select() == ENGINE_DIFFUSERS
@pytest.mark.parametrize(
"backend, expected",
[("rocm", "rocm"), ("cuda", "cuda"), ("xpu", "vulkan"), ("cpu", "auto"), ("mps", "auto")],
)
def test_install_accelerator_maps_backend(backend, expected):
assert r._install_accelerator_for(backend) == expected
def test_force_native_install_uses_gpu_accelerator(monkeypatch):
# Forcing sd_cpp on a ROCm host with no binary must install the ROCm build, not the
# default CPU one -- otherwise the forced-native generation silently runs on CPU.
_set_device(monkeypatch, "rocm")
_set_runnable(monkeypatch)
seen = {}
def _fake_ensure(**kwargs):
seen.update(kwargs)
return "/usr/bin/sd-cli"
monkeypatch.setattr(r, "ensure_sd_cpp_binary", _fake_ensure)
monkeypatch.setenv("UNSLOTH_DIFFUSION_ENGINE", "sd_cpp")
assert _select() == ENGINE_SD_CPP
assert seen.get("accelerator") == "rocm"
# ── active_status annotation ──────────────────────────────────────────────────
def test_active_status_injects_engine_and_reason(monkeypatch):
_set_device(monkeypatch, "cpu")
_set_binary(monkeypatch, None)
_select() # -> diffusers fallback (no binary)
st = r.active_status()
assert st["engine"] == ENGINE_DIFFUSERS
assert st["fallback_reason"] and "binary unavailable" in st["fallback_reason"]