Memory planning and dense-quant path: size a local diffusers base's resident companions from its on-disk VAE and text-encoder weights instead of folding them to zero, feed the distilled variant hint into the runtime headroom estimate so turbo and schnell models are not over-reserved, place group-offload companions resident before attaching the transformer hooks so a failed placement falls back to whole-module offload instead of crashing, and bail out of the dense transformer download before it starts when the requested quant scheme is unsupported so the load falls back to GGUF cleanly. sd.cpp stack: scrub the native path lease secret from sd-cli child env, redact native load-progress errors, forward the resolved accelerator when auto-installing a forced-native binary, release stale diffusion GPU ownership on CPU-native loads, and remove the sd.cpp install tree on uninstall. Prequant and scripts: reject prequant artifacts missing base_model_id when a base is requested, expanduser before checkpoint existence checks, record and validate the int8 exclusion filter and fp8 fast-accum in checkpoint metadata, make verify_prequant_backend allowlist its local checkpoint and fail on missing or bad LPIPS and on load-peak regressions, average only finite PSNR values in diffusion_quality, and reset the process-wide attention backend between perf probe variants. API and UI: normalize attention_backend casing before Literal validation, close hidden popovers when leaving the Images page, and clear the stale quant label when loading a direct local GGUF file.
182 lines
6.8 KiB
Python
182 lines
6.8 KiB
Python
# SPDX-License-Identifier: AGPL-3.0-only
|
|
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
|
|
|
|
"""Tests for the diffusion engine router (diffusers vs native sd.cpp selection)."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from types import SimpleNamespace
|
|
|
|
import pytest
|
|
|
|
from core.inference import diffusion_engine_router as r
|
|
from core.inference.diffusion_families import detect_family
|
|
from core.inference.sd_cpp_engine import ENGINE_DIFFUSERS, ENGINE_SD_CPP
|
|
|
|
_ENVS = (
|
|
"UNSLOTH_DIFFUSION_ENGINE",
|
|
"UNSLOTH_DIFFUSION_SD_CPP",
|
|
"UNSLOTH_DIFFUSION_SD_CPP_MPS",
|
|
"UNSLOTH_DIFFUSION_SD_CPP_INSTALL",
|
|
)
|
|
|
|
|
|
@pytest.fixture(autouse = True)
|
|
def _clean_env_and_state(monkeypatch):
|
|
for e in _ENVS:
|
|
monkeypatch.delenv(e, raising = False)
|
|
# A light status-capable stub so neither selection nor active_status() imports the
|
|
# heavy diffusers/sd.cpp backends; the active engine NAME comes from module state.
|
|
monkeypatch.setattr(
|
|
r,
|
|
"get_active_diffusion_engine",
|
|
lambda: SimpleNamespace(status = lambda: {"loaded": False, "repo_id": None}),
|
|
)
|
|
yield
|
|
|
|
|
|
def _set_device(monkeypatch, backend):
|
|
monkeypatch.setattr(
|
|
r,
|
|
"resolve_diffusion_device_target",
|
|
lambda: SimpleNamespace(backend = backend, device = backend),
|
|
)
|
|
|
|
|
|
def _set_binary(monkeypatch, path):
|
|
monkeypatch.setattr(r, "ensure_sd_cpp_binary", lambda **_: path)
|
|
|
|
|
|
def _set_runnable(monkeypatch, version = "sd-cli v0"):
|
|
"""Stub the runnability probe so a stubbed binary path is treated as executable
|
|
(the router now probes ``SdCppEngine(...).version()`` before committing to native)."""
|
|
monkeypatch.setattr(r, "SdCppEngine", lambda **_: SimpleNamespace(version = lambda: version))
|
|
|
|
|
|
def _select(fam_name = "z-image"):
|
|
"""Activate the engine for a family and return which engine was chosen."""
|
|
r.select_and_activate_engine(detect_family(fam_name))
|
|
return r.active_engine_name()
|
|
|
|
|
|
# ── core selection matrix ─────────────────────────────────────────────────────
|
|
|
|
|
|
def test_cpu_with_binary_and_supported_family_picks_sd_cpp(monkeypatch):
|
|
_set_device(monkeypatch, "cpu")
|
|
_set_binary(monkeypatch, "/usr/bin/sd-cli")
|
|
_set_runnable(monkeypatch)
|
|
assert _select() == ENGINE_SD_CPP
|
|
assert r.active_engine_name() == ENGINE_SD_CPP
|
|
|
|
|
|
def test_present_but_not_runnable_binary_falls_back(monkeypatch):
|
|
# A binary that exists but cannot run (version() -> None) must fall back to
|
|
# diffusers at selection, not commit native and fail inside the load.
|
|
_set_device(monkeypatch, "cpu")
|
|
_set_binary(monkeypatch, "/usr/bin/sd-cli")
|
|
monkeypatch.setattr(r, "SdCppEngine", lambda **_: SimpleNamespace(version = lambda: None))
|
|
assert _select() == ENGINE_DIFFUSERS
|
|
assert "binary unavailable" in (r.active_status()["fallback_reason"] or "")
|
|
|
|
|
|
@pytest.mark.parametrize("gpu", ["cuda", "rocm", "xpu"])
|
|
def test_gpu_backends_use_diffusers(monkeypatch, gpu):
|
|
_set_device(monkeypatch, gpu)
|
|
_set_binary(monkeypatch, "/usr/bin/sd-cli") # even with a binary, GPU stays diffusers
|
|
assert _select() == ENGINE_DIFFUSERS
|
|
assert "uses diffusers" in (r.active_status()["fallback_reason"] or "")
|
|
|
|
|
|
def test_forced_diffusers_overrides_cpu(monkeypatch):
|
|
_set_device(monkeypatch, "cpu")
|
|
_set_binary(monkeypatch, "/usr/bin/sd-cli")
|
|
monkeypatch.setenv("UNSLOTH_DIFFUSION_ENGINE", "diffusers")
|
|
assert _select() == ENGINE_DIFFUSERS
|
|
assert "forced" in (r.active_status()["fallback_reason"] or "")
|
|
|
|
|
|
def test_sd_cpp_disabled_uses_diffusers(monkeypatch):
|
|
_set_device(monkeypatch, "cpu")
|
|
_set_binary(monkeypatch, "/usr/bin/sd-cli")
|
|
monkeypatch.setenv("UNSLOTH_DIFFUSION_SD_CPP", "0")
|
|
assert _select() == ENGINE_DIFFUSERS
|
|
assert "disabled" in (r.active_status()["fallback_reason"] or "")
|
|
|
|
|
|
def test_mps_default_diffusers_but_optin_sd_cpp(monkeypatch):
|
|
_set_device(monkeypatch, "mps")
|
|
_set_binary(monkeypatch, "/usr/bin/sd-cli")
|
|
_set_runnable(monkeypatch)
|
|
# Default: MPS is not native-eligible -> diffusers.
|
|
assert _select() == ENGINE_DIFFUSERS
|
|
# Opt in: MPS routes to sd.cpp.
|
|
monkeypatch.setenv("UNSLOTH_DIFFUSION_SD_CPP_MPS", "1")
|
|
assert _select() == ENGINE_SD_CPP
|
|
|
|
|
|
def test_unsupported_family_falls_back(monkeypatch):
|
|
_set_device(monkeypatch, "cpu")
|
|
_set_binary(monkeypatch, "/usr/bin/sd-cli")
|
|
monkeypatch.setattr(r, "family_sd_cpp_supported", lambda fam: False)
|
|
assert _select() == ENGINE_DIFFUSERS
|
|
assert "no native sd.cpp asset mapping" in (r.active_status()["fallback_reason"] or "")
|
|
|
|
|
|
def test_missing_binary_falls_back(monkeypatch):
|
|
_set_device(monkeypatch, "cpu")
|
|
_set_binary(monkeypatch, None) # install unavailable
|
|
assert _select() == ENGINE_DIFFUSERS
|
|
assert "binary unavailable" in (r.active_status()["fallback_reason"] or "")
|
|
|
|
|
|
def test_force_sd_cpp_on_gpu_when_binary_present(monkeypatch):
|
|
_set_device(monkeypatch, "cuda")
|
|
_set_binary(monkeypatch, "/usr/bin/sd-cli")
|
|
_set_runnable(monkeypatch)
|
|
monkeypatch.setenv("UNSLOTH_DIFFUSION_ENGINE", "sd_cpp")
|
|
assert _select() == ENGINE_SD_CPP
|
|
|
|
|
|
def test_force_sd_cpp_without_binary_falls_back(monkeypatch):
|
|
_set_device(monkeypatch, "cuda")
|
|
_set_binary(monkeypatch, None)
|
|
monkeypatch.setenv("UNSLOTH_DIFFUSION_ENGINE", "sd_cpp")
|
|
assert _select() == ENGINE_DIFFUSERS
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"backend, expected",
|
|
[("rocm", "rocm"), ("cuda", "cuda"), ("xpu", "vulkan"), ("cpu", "auto"), ("mps", "auto")],
|
|
)
|
|
def test_install_accelerator_maps_backend(backend, expected):
|
|
assert r._install_accelerator_for(backend) == expected
|
|
|
|
|
|
def test_force_native_install_uses_gpu_accelerator(monkeypatch):
|
|
# Forcing sd_cpp on a ROCm host with no binary must install the ROCm build, not the
|
|
# default CPU one -- otherwise the forced-native generation silently runs on CPU.
|
|
_set_device(monkeypatch, "rocm")
|
|
_set_runnable(monkeypatch)
|
|
seen = {}
|
|
|
|
def _fake_ensure(**kwargs):
|
|
seen.update(kwargs)
|
|
return "/usr/bin/sd-cli"
|
|
|
|
monkeypatch.setattr(r, "ensure_sd_cpp_binary", _fake_ensure)
|
|
monkeypatch.setenv("UNSLOTH_DIFFUSION_ENGINE", "sd_cpp")
|
|
assert _select() == ENGINE_SD_CPP
|
|
assert seen.get("accelerator") == "rocm"
|
|
|
|
|
|
# ── active_status annotation ──────────────────────────────────────────────────
|
|
|
|
|
|
def test_active_status_injects_engine_and_reason(monkeypatch):
|
|
_set_device(monkeypatch, "cpu")
|
|
_set_binary(monkeypatch, None)
|
|
_select() # -> diffusers fallback (no binary)
|
|
st = r.active_status()
|
|
assert st["engine"] == ENGINE_DIFFUSERS
|
|
assert st["fallback_reason"] and "binary unavailable" in st["fallback_reason"]
|