From 4cc35aa87a740ff77462820b061548a57ae3066e Mon Sep 17 00:00:00 2001 From: Daniel Han Date: Sun, 12 Jul 2026 12:21:14 +0000 Subject: [PATCH] Tighten comments in the image stack tests and scripts --- scripts/build_prequant_checkpoint.py | 25 +- scripts/diffusion_bench.py | 12 +- scripts/diffusion_quality.py | 13 +- scripts/fbcache_flux_probe.py | 5 +- scripts/fp8_overflow_check.py | 6 +- scripts/perf_levers_probe.py | 12 +- scripts/verify_prequant_backend.py | 10 +- scripts/video_quality.py | 12 +- .../backend/tests/test_diffusion_backend.py | 277 +++++++++--------- studio/backend/tests/test_video_backend.py | 181 ++++++------ studio/install_sd_cpp_prebuilt.py | 49 ++-- 11 files changed, 290 insertions(+), 312 deletions(-) diff --git a/scripts/build_prequant_checkpoint.py b/scripts/build_prequant_checkpoint.py index 9766467351..96f04b039a 100644 --- a/scripts/build_prequant_checkpoint.py +++ b/scripts/build_prequant_checkpoint.py @@ -83,17 +83,16 @@ def main(argv = None) -> int: args.base, subfolder = "transformer", torch_dtype = torch.bfloat16, token = args.hf_token ).to("cuda") print(f" quantising in place ({scheme}) ...", flush = True) - # Mirror the runtime path EXACTLY (the offline == runtime, LPIPS-0 invariant): for int8 also - # skip the M=1 AdaLN-modulation / conditioning-embedder projections, else the saved checkpoint - # bakes them as int8 and crashes (torch._int_mm needs M>16) at the first denoise step on - # Flux / Qwen. fp8 / fp4 / mx use scaled_mm (no M limit) -> exclude_tokens_for_scheme returns (). + # Mirror the runtime path EXACTLY (offline == runtime, LPIPS-0 invariant): for int8 also skip + # the M=1 AdaLN-modulation / conditioning-embedder projections, else the checkpoint bakes them + # as int8 and crashes (torch._int_mm needs M>16) at the first denoise step on Flux / Qwen. fp8 + # / fp4 / mx use scaled_mm (no M limit) -> exclude_tokens_for_scheme returns (). exclude_name_tokens = exclude_tokens_for_scheme(scheme) # fp8 and mxfp8 assert a bf16 weight, so their filter must skip any non-bf16 Linear the - # transformer keeps: a mixed-precision DiT (Wan / Hunyuan) retains its _keep_in_fp32_modules in - # fp32 even under torch_dtype=bf16, so quantising one would raise inside quantize_ and abort the - # whole pass. nvfp4 quantises fp32 fine, so it is not gated. Runtime quantize_transformer gates - # this on scheme membership; mirror it here so the offline checkpoint quantises the exact same - # layer set (offline == runtime). + # transformer keeps: a mixed-precision DiT (Wan / Hunyuan) keeps its _keep_in_fp32_modules in + # fp32 even under torch_dtype=bf16, so quantising one raises inside quantize_ and aborts the + # pass. nvfp4 quantises fp32 fine, so it isn't gated. Runtime quantize_transformer gates on + # scheme membership; mirror it here so the offline checkpoint quantises the same layer set. require_bf16 = scheme in _REQUIRE_BF16_SCHEMES # fp8 bakes the accumulate mode into the saved kernels; record the resolved choice so the # loader can refuse a checkpoint whose baked value contradicts an explicit runtime request. @@ -118,10 +117,10 @@ def main(argv = None) -> int: "family": fam.name, "scheme": scheme, "min_features": args.min_features, - # The layers skipped for this scheme (int8's M=1 modulation projections; () for - # the scaled_mm schemes), whether non-bf16 Linears were skipped (the scaled_mm - # bf16 gate), and, for fp8, the baked accumulate mode. All let the loader reject a - # checkpoint that would not match the runtime path. + # The layers skipped for this scheme (int8's M=1 modulation projections; () for the + # scaled_mm schemes), whether non-bf16 Linears were skipped (the scaled_mm bf16 gate), and, + # for fp8, the baked accumulate mode. All let the loader reject a checkpoint that wouldn't + # match the runtime path. "exclude_name_tokens": list(exclude_name_tokens), "require_bf16": require_bf16, "fast_accum": fast_accum, diff --git a/scripts/diffusion_bench.py b/scripts/diffusion_bench.py index 43a92ac8ee..f27ad17aeb 100644 --- a/scripts/diffusion_bench.py +++ b/scripts/diffusion_bench.py @@ -43,8 +43,8 @@ from typing import Any, Optional # The backend package lives at unsloth/studio/backend; this file is at # unsloth/scripts/diffusion_bench.py. Put the backend root on sys.path so -# ``core.inference.diffusion`` imports the same way the server does. (The actual -# backend import is deferred into main() so --help never triggers torch.) +# ``core.inference.diffusion`` imports as the server does. (The backend import is deferred +# into main() so --help never triggers torch.) _BACKEND_ROOT = Path(__file__).resolve().parent.parent / "studio" / "backend" if str(_BACKEND_ROOT) not in sys.path: sys.path.insert(0, str(_BACKEND_ROOT)) @@ -370,10 +370,10 @@ def _compare(args: argparse.Namespace) -> int: print(" refusing noisy comparison (pass --force-compare to override).", flush = True) return 2 - # PSNR vs the stored reference image. The baseline stores an absolute reference_png, - # which breaks if the baseline directory was copied/moved, so fall back to reference.png - # next to the baseline JSON. A still-missing reference is a failure below, not a silent - # pass -- otherwise the benchmark would report PASS having done no image comparison. + # PSNR vs the stored reference image. The baseline stores an absolute reference_png, which + # breaks if the baseline dir was copied/moved, so fall back to reference.png next to the + # baseline JSON. A still-missing reference is a failure below, not a silent pass -- else the + # benchmark reports PASS having done no image comparison. ref_png = Path(baseline.get("accuracy", {}).get("reference_png", "")) if not ref_png.is_file(): ref_png = baseline_path.parent / "reference.png" diff --git a/scripts/diffusion_quality.py b/scripts/diffusion_quality.py index a8d504713d..ca121304d3 100644 --- a/scripts/diffusion_quality.py +++ b/scripts/diffusion_quality.py @@ -67,9 +67,9 @@ def _to_rgb(path_or_img: Any) -> Any: return np.asarray(img.convert("RGB"), dtype = np.float64) -# Finite PSNR (dB) a perfect (inf) sample is capped to when averaged with imperfect ones, -# so a lossless render counts as excellent without hiding diverged samples. Well above the -# ~37 dB compile and ~21 dB quant noise floors this harness reports. +# Finite PSNR (dB) a perfect (inf) sample is capped to when averaged with imperfect ones, so a +# lossless render counts as excellent without hiding diverged samples. Well above the ~37 dB +# compile and ~21 dB quant noise floors this harness reports. _PERFECT_MATCH_PSNR = 100.0 @@ -287,10 +287,9 @@ def _compare( def _mean(xs: list[float]) -> Optional[float]: # +inf marks an identical render (reference vs itself, or a lossless quant/offload) - # scoring PSNR=inf -- the case this harness verifies. Report inf ONLY when every - # sample is inf; a mix of inf and finite means some renders diverged, so a bare inf - # would mask those bad samples. Cap the perfect ones to a high finite PSNR and - # average so the drift still shows. (Only PSNR is ever inf; SSIM/CLIP stay finite.) + # scoring PSNR=inf -- the case this harness verifies. Report inf ONLY when every sample is + # inf; a mix means some renders diverged, so a bare inf would mask them. Cap the perfect + # ones to a high finite PSNR and average so the drift still shows. (Only PSNR is ever inf.) if not xs: return None if all(x == math.inf for x in xs): diff --git a/scripts/fbcache_flux_probe.py b/scripts/fbcache_flux_probe.py index c0462f1947..d323e837aa 100644 --- a/scripts/fbcache_flux_probe.py +++ b/scripts/fbcache_flux_probe.py @@ -102,9 +102,8 @@ def run( apply_first_block_cache(pipe.transformer, FirstBlockCacheConfig(threshold = threshold)) if compile_: # FBCache's per-step decision is a graph break, so a cached run must compile with - # fullgraph=False (mirroring the production path); fullgraph=True would fail the - # warmup compile and the row would silently fall back to an eager cached run, - # producing misleading speedup numbers. + # fullgraph=False (mirroring production); fullgraph=True would fail the warmup compile and + # the row would fall back to an eager cached run, producing misleading speedups. fullgraph = threshold is None try: pipe.transformer.compile_repeated_blocks(fullgraph = fullgraph, dynamic = True) diff --git a/scripts/fp8_overflow_check.py b/scripts/fp8_overflow_check.py index 157b9cce30..b36cc6e93b 100644 --- a/scripts/fp8_overflow_check.py +++ b/scripts/fp8_overflow_check.py @@ -63,9 +63,9 @@ def _run(fast_accum, steps, res, seed, mf): if m > stats["max_abs"]: stats["max_abs"] = m - # Hook the quantised linears (where an fp8-accumulation overflow would surface). - # Run EAGER: forward hooks don't trace through torch.compile, and the fp8 fast-accum - # accumulation is identical compiled or eager -- compile only changes scheduling. + # Hook the quantised linears (where an fp8-accumulation overflow would surface). Run EAGER: + # forward hooks don't trace through torch.compile, and the fp8 fast-accum accumulation is + # identical compiled or eager -- compile only changes scheduling. for m in pipe.transformer.modules(): if isinstance(m, nn.Linear): m.register_forward_hook(hook) diff --git a/scripts/perf_levers_probe.py b/scripts/perf_levers_probe.py index 565e28c6b9..16f8d826c5 100644 --- a/scripts/perf_levers_probe.py +++ b/scripts/perf_levers_probe.py @@ -36,9 +36,9 @@ def _lpips(ref, arr): import lpips import torch - # Keep the metric model on CPU: caching it on CUDA leaves it resident across - # variants, and each run resets peak-memory stats, so its VRAM would be charged - # to (and reduce headroom for) every later variant's measurement. + # Keep the metric model on CPU: caching it on CUDA leaves it resident across variants, and + # each run resets peak-memory stats, so its VRAM would be charged to (and reduce headroom + # for) every later variant's measurement. if _LP["fn"] is None: _LP["fn"] = lpips.LPIPS(net = "alex", verbose = False).eval() @@ -150,9 +150,9 @@ def run( return None else: # set_attention_backend pins diffusers' PROCESS-WIDE active backend, and a fresh - # transformer's processors (backend None) inherit it. Force native for the no-attn - # variants so they aren't silently measured under a prior variant's kernel (e.g. - # fbcache running with a leftover sage backend). + # transformer's processors (backend None) inherit it. Force native for the no-attn variants + # so they aren't measured under a prior variant's kernel (e.g. fbcache with a leftover sage + # backend). try: pipe.transformer.set_attention_backend("native") except Exception as exc: # noqa: BLE001 — best-effort isolation diff --git a/scripts/verify_prequant_backend.py b/scripts/verify_prequant_backend.py index 66dc89b0c6..d0c2518c89 100644 --- a/scripts/verify_prequant_backend.py +++ b/scripts/verify_prequant_backend.py @@ -38,9 +38,9 @@ logging.basicConfig(level = logging.INFO, format = "%(message)s") LOGGER = logging.getLogger("verify_prequant") # Prequant and runtime produce the SAME quantized weights, so their images must be -# near-identical; anything above this LPIPS means the prequant path diverged and the run -# fails. The prequant load peak must also sit clearly below the dense runtime peak (the -# whole point of the path); require at least this fractional headroom. +# near-identical; anything above this LPIPS means the prequant path diverged and the run fails. +# The prequant load peak must also sit clearly below the dense runtime peak (the point of the +# path); require at least this fractional headroom. LPIPS_MAX = 0.02 PREQUANT_PEAK_MAX_FRACTION = 0.75 _RUNTIME_PEAK_FILE = OUT / "runtime_peak.txt" @@ -103,8 +103,8 @@ def run(mode, steps, seed, res): if mode == "prequant": # A local checkpoint is refused unless its directory is allowlisted (unpickling an - # arbitrary file is unsafe). This verifier's CKPT is operator-supplied and trusted, - # so allowlist its directory here or the load returns None and measures nothing. + # arbitrary file is unsafe). This verifier's CKPT is operator-supplied and trusted, so + # allowlist its directory here or the load returns None and measures nothing. ckpt_dir = os.path.dirname(os.path.realpath(CKPT)) existing = os.environ.get(ALLOW_LOCAL_PREQUANT_PATH_ENV, "") os.environ[ALLOW_LOCAL_PREQUANT_PATH_ENV] = ( diff --git a/scripts/video_quality.py b/scripts/video_quality.py index 21234a08d0..40d3732f2c 100644 --- a/scripts/video_quality.py +++ b/scripts/video_quality.py @@ -159,13 +159,11 @@ def clip_metrics( """All frame metrics for one candidate clip vs the reference clip.""" import numpy as np - # The gate holds the requested shape (num_frames) fixed for reference and - # candidate alike, so both clips must decode to the same frame count. A - # shorter candidate is a truncated/corrupt render, not a valid one; comparing - # only the shared prefix would let good early frames mask the missing tail, so - # the mismatch is recorded and gated as FAIL (see verdict()) rather than - # silently dropped. No off-by-one is tolerated: nothing in the encode/decode - # path justifies one. + # The gate holds the requested shape (num_frames) fixed for reference and candidate, so both + # clips must decode to the same frame count. A shorter candidate is a truncated/corrupt render; + # comparing only the shared prefix would let good early frames mask the missing tail, so the + # mismatch is recorded and gated as FAIL (see verdict()) rather than dropped. No off-by-one is + # tolerated. ref_count, cand_count = len(ref_frames), len(cand_frames) frame_count_mismatch = ref_count != cand_count n = min(ref_count, cand_count) diff --git a/studio/backend/tests/test_diffusion_backend.py b/studio/backend/tests/test_diffusion_backend.py index 35cbaab0e1..ddbf79e23c 100644 --- a/studio/backend/tests/test_diffusion_backend.py +++ b/studio/backend/tests/test_diffusion_backend.py @@ -25,11 +25,10 @@ from core.inference.diffusion import ( _resolve_diffusion_compute_dtype, ) -# diffusion.py imports the compile/arch patch modules LAZILY (they pull torch at module -# level, and diffusion.py must stay importable on a torchless native install). Import them -# here at collection time -- under the real torch -- so they are cached in sys.modules -# before the fake-torch fixtures swap it out; otherwise the lazy import inside load_pipeline -# would try to build them against the incomplete stub torch. +# diffusion.py imports the compile/arch patch modules LAZILY (they pull torch at module level, +# and diffusion.py must stay importable on a torchless native install). Import them here at +# collection time -- under the real torch -- so they're cached in sys.modules before the +# fake-torch fixtures swap it out; else the lazy import would build them against the stub torch. import core.inference.diffusion_eager_patches # noqa: E402,F401 import core.inference.diffusion_arch_patches # noqa: E402,F401 from core.inference.diffusion_families import ( @@ -44,9 +43,9 @@ from core.inference.diffusion_families import ( def test_clamp_max_side_bounds_oversized_init(): - # img2img / inpaint derive the OUTPUT size from the uploaded image; an oversized upload - # (up to the 4096/side decode cap = 4x the txt2img 2048 ceiling) would drive an OOM-scale - # latent. _clamp_max_side bounds the longest side to 2048, preserving aspect ratio. + # img2img / inpaint derive OUTPUT size from the uploaded image; an oversized upload (up to + # the 4096/side decode cap = 4x the txt2img 2048 ceiling) drives an OOM-scale latent. + # _clamp_max_side bounds the longest side to 2048, preserving aspect ratio. from PIL import Image # A 12MP-shaped landscape photo -> longest side clamped to 2048, 4:3 aspect preserved. @@ -112,9 +111,9 @@ def test_detect_family_from_repo_id(): def test_detect_family_matches_reject_and_alias_by_segment(): - # Reject keywords and short aliases must match whole path/name segments, not raw - # substrings, so an unrelated word that merely CONTAINS one does not misroute a - # valid base model (regression: substring matching broke these). + # Reject keywords and short aliases must match whole path/name segments, not raw substrings, + # so an unrelated word that CONTAINS one doesn't misroute a valid base model (regression: + # substring matching broke these). assert detect_family("/models/edited/z-image-turbo-Q4_K_M.gguf").name == "z-image" assert detect_family("unsloth/Z-Image-Edition-GGUF").name == "z-image" assert detect_family("/models/kontextual/z-image-turbo-Q4_K_M.gguf").name == "z-image" @@ -129,9 +128,9 @@ def test_detect_family_matches_reject_and_alias_by_segment(): def test_detect_family_edit_keyword_scoped_to_basename(): from core.inference.diffusion_families import detect_family_for_pick - # A parent directory named `edit`/`inpaint` must NOT poison a valid pick: only - # the model id / filename basename is scanned for reject keywords. A direct - # local pick arrives as (parent_dir, filename). + # A parent directory named `edit`/`inpaint` must NOT poison a valid pick: only the model id + # / filename basename is scanned for reject keywords. A direct local pick arrives as + # (parent_dir, filename). assert detect_family("/models/edit") is None # the dir alone is ambiguous assert detect_family_for_pick("/models/edit", "Z-Image-Turbo-Q4.gguf").name == "z-image" assert detect_family_for_pick("/models/inpaint", "qwen-image-2512-Q4.gguf").name == "qwen-image" @@ -246,9 +245,9 @@ class _FakePipe: def enable_vae_slicing(self) -> None: self.vae_sliced = True - # Explicit signature (not just **kwargs) so generate()'s signature-gated - # guards for negative_prompt / callback_on_step_end actually take effect — - # a **kwargs-only fake would make `"negative_prompt" in signature` always False. + # Explicit signature (not just **kwargs) so generate()'s signature-gated guards for + # negative_prompt / callback_on_step_end take effect -- a **kwargs-only fake would make + # `"negative_prompt" in signature` always False. def __call__( self, *, @@ -401,14 +400,14 @@ def fake_runtime(monkeypatch): diffusers.QwenImageInpaintPipeline = _FakeInpaintPipeline # Instruction-editing pipeline (Qwen-Image-Edit): its own pipeline IS the loaded one. diffusers.QwenImageEditPlusPipeline = _FakePipeline - # Ideogram 4, so its guidance_scale/guidance_schedule pairing is exercisable. It loads - # only as a full pipeline (two DiTs), assembled per-component by load_ideogram4_pipeline - # -- stub that to a fake pipe so the guidance path is reachable without real weights. + # Ideogram 4, so its guidance_scale/guidance_schedule pairing is exercisable. It loads only + # as a full pipeline (two DiTs), assembled per-component by load_ideogram4_pipeline -- stub + # that to a fake pipe so the guidance path is reachable without real weights. diffusers.Ideogram4Pipeline = _FakePipeline diffusers.Ideogram4Transformer2DModel = _FakeTransformer - # SDXL: a U-Net family. Its single-file checkpoint is the whole pipeline, so the - # pipeline class carries from_single_file; UNet2DConditionModel is the denoiser - # class (fetched but unused on the pipeline/single-file-pipeline paths). + # SDXL: a U-Net family. Its single-file checkpoint is the whole pipeline, so the pipeline + # class carries from_single_file; UNet2DConditionModel is the denoiser class (fetched but + # unused on the pipeline/single-file-pipeline paths). diffusers.StableDiffusionXLPipeline = _FakePipeline diffusers.UNet2DConditionModel = _FakeTransformer diffusers.StableDiffusionXLImg2ImgPipeline = _FakeImg2ImgPipeline @@ -484,9 +483,9 @@ def test_load_generate_unload_gguf(fake_runtime, tmp_path): def test_dense_speed_auto_defers_compile_to_third_generation(fake_runtime, tmp_path, monkeypatch): - # Dense models with speed unset stay bit-identical eager for the first two - # generations; the 3rd engages the `default` profile mid-session (repeated - # use amortises the one-time compile), upgrading attention alongside it. + # Dense models with speed unset stay bit-identical eager for the first two generations; the + # 3rd engages the `default` profile mid-session (repeated use amortises the one-time compile), + # upgrading attention alongside it. from core.inference import diffusion as dmod monkeypatch.setattr(dmod, "compile_eligible", lambda *a, **k: True) @@ -542,10 +541,10 @@ def test_dense_speed_auto_defers_compile_to_third_generation(fake_runtime, tmp_p def test_deferred_speed_skips_when_lora_requested(fake_runtime, tmp_path, monkeypatch): - # A compiled transformer rejects LoRA (supports_lora is False once compiled), and _apply_loras - # raises before its unchanged-selection no-op, so engaging the deferred compile on a generation - # that requests a LoRA would permanently break every LoRA generation on this load. The deferral - # must skip while a LoRA is requested and engage only on a later LoRA-free generation. + # A compiled transformer rejects LoRA (supports_lora False once compiled), and _apply_loras + # raises before its unchanged-selection no-op, so engaging the deferred compile on a LoRA + # generation would permanently break every LoRA generation on this load. The deferral must + # skip while a LoRA is requested and engage only on a later LoRA-free generation. from core.inference import diffusion as dmod monkeypatch.setattr(dmod, "compile_eligible", lambda *a, **k: True) @@ -578,10 +577,10 @@ def test_deferred_speed_skips_when_lora_requested(fake_runtime, tmp_path, monkey def test_deferred_speed_skips_while_adapter_attached(fake_runtime, tmp_path, monkeypatch): - # Even a generation that requests NO LoRA must defer the compile while an adapter from a PRIOR - # generation is still attached: _apply_loras runs AFTER the engage, so compiling here would bake - # the resident adapter into the graph and the subsequent unload (swallowed on a compiled pipe) - # would leave it active forever -- silent wrong output. Defer until _apply_loras clears it. + # Even a NO-LoRA generation must defer the compile while an adapter from a PRIOR generation is + # still attached: _apply_loras runs AFTER the engage, so compiling here would bake the resident + # adapter into the graph and the later unload (swallowed on a compiled pipe) would leave it + # active forever -- silent wrong output. Defer until _apply_loras clears it. from core.inference import diffusion as dmod monkeypatch.setattr(dmod, "compile_eligible", lambda *a, **k: True) @@ -620,10 +619,10 @@ def test_deferred_speed_skips_while_adapter_attached(fake_runtime, tmp_path, mon def test_deferred_speed_preserves_explicit_attention(fake_runtime, tmp_path, monkeypatch): - # A dense model loaded with Speed left on Auto but Attention explicitly pinned - # (e.g. "native" to avoid cuDNN) must KEEP that choice when the 3rd generation - # engages the deferred `default` profile. The auto cuDNN upgrade only applies when - # attention was left on auto -- never when the caller pinned a backend. + # A dense model loaded with Speed on Auto but Attention explicitly pinned (e.g. "native" to + # avoid cuDNN) must KEEP that choice when the 3rd generation engages the deferred `default` + # profile. The auto cuDNN upgrade applies only when attention was left on auto, never when the + # caller pinned a backend. from core.inference import diffusion as dmod monkeypatch.setattr(dmod, "compile_eligible", lambda *a, **k: True) @@ -634,9 +633,9 @@ def test_deferred_speed_preserves_explicit_attention(fake_runtime, tmp_path, mon ) monkeypatch.setattr(dmod, "apply_attention_backend", lambda pipe, backend, logger = None: backend) - # A select mock that -- unlike a bare "auto -> cuDNN" stub -- HONORS an explicit - # request: "native" stays on the default (None) even under a speed profile, and only - # a left-unset ("auto"/None) request upgrades to cuDNN when speed is active. + # A select mock that -- unlike a bare "auto -> cuDNN" stub -- HONORS an explicit request: + # "native" stays on the default (None) even under a speed profile, and only a left-unset + # ("auto"/None) request upgrades to cuDNN when speed is active. def fake_select( target, requested, @@ -1282,9 +1281,9 @@ def test_load_sdxl_rejects_untrusted_repo(fake_runtime): def test_validate_gates_untrusted_base_repo(fake_runtime, tmp_path): - # A companion base_repo also loads via from_pretrained, so a trusted GGUF model_path must - # not smuggle in an arbitrary remote base: base_repo clears the same trust bar as a non-GGUF - # repo id (mirrors the video loader), and the check runs before any GPU handoff. + # A companion base_repo also loads via from_pretrained, so a trusted GGUF model_path must not + # smuggle in an arbitrary remote base: base_repo clears the same trust bar as a non-GGUF repo + # id (mirrors the video loader), and the check runs before any GPU handoff. backend = DiffusionBackend() with pytest.raises(ValueError, match = "base_repo"): backend.validate_load_request( @@ -1295,8 +1294,8 @@ def test_validate_gates_untrusted_base_repo(fake_runtime, tmp_path): ) # A local base_repo dir that is NOT a diffusers pipeline (no model_index.json) is rejected # HERE, before the GPU handoff: it passes the any-existing-path trust check but the base loads - # via from_pretrained (needs model_index.json), so it would otherwise evict the resident model - # and only then fail in the background load. + # via from_pretrained (needs model_index.json), so it would else evict the resident model and + # only then fail in the background load. bad_base = tmp_path / "bare-base" bad_base.mkdir() with pytest.raises(ValueError, match = "model_index.json"): @@ -1344,7 +1343,7 @@ def test_resolve_local_single_file(tmp_path): # A PEFT LoRA adapter folder (adapter_config.json + adapter_model.safetensors), even with a # family-token name, is NOT a base checkpoint: from_single_file would fail on the adapter - # weights AFTER the route evicted the resident GPU model, so it must not be reinterpreted as a + # weights AFTER the route evicted the resident model, so it must not be reinterpreted as a # single_file pick -> None (the pipeline pick then 400s in validation, before the handoff). adapter = tmp_path / "flux-style-lora" adapter.mkdir() @@ -1359,10 +1358,10 @@ def test_resolve_local_single_file(tmp_path): def test_resolve_base_repo_drops_untrusted_card_tag(monkeypatch): - # When no base_repo is passed, the base is resolved from the GGUF repo's base_model card - # tag -- attacker-controlled metadata on any remote repo -- and then loaded via - # from_pretrained. An untrusted tag must be dropped in favour of the curated family default, - # so an attacker GGUF repo cannot point the base at an arbitrary repo to be deserialized. + # With no base_repo, the base is resolved from the GGUF repo's base_model card tag -- + # attacker-controlled metadata on any remote repo -- then loaded via from_pretrained. An + # untrusted tag must be dropped for the curated family default, so an attacker GGUF repo can't + # point the base at an arbitrary repo to be deserialized. import core.inference.diffusion as dmod fam = detect_family("unsloth/FLUX.1-dev-GGUF") @@ -1455,10 +1454,10 @@ def test_generate_without_load_raises(fake_runtime): def test_failed_load_restores_backend_flags(fake_runtime, tmp_path, monkeypatch): - # A failure AFTER apply_speed_optims (here an OOM in apply_memory_plan) must go - # through the load's try/finally and restore the process-global TF32 / cudnn flags, - # so a later `off` load is still bit-identical, and must not commit a partial state. - # Regression: a refactor dropped this guard, leaking the flags on a failed load. + # A failure AFTER apply_speed_optims (here an OOM in apply_memory_plan) must go through the + # load's try/finally and restore the process-global TF32 / cudnn flags, so a later `off` load + # is still bit-identical, and must not commit partial state. Regression: a refactor dropped + # this guard, leaking the flags on a failed load. (tmp_path / "model.gguf").write_bytes(b"x") backend = DiffusionBackend() @@ -1619,9 +1618,9 @@ def _load_ideogram(backend, tmp_path): def test_ideogram_rejects_single_file_and_gguf_kinds(fake_runtime, tmp_path): - # Ideogram 4 needs two DiTs assembled per-component, so there is no transformer-only - # single-file or GGUF load: the explicit kinds must be rejected up front (before a - # load evicts a working model), not assembled into a pipeline missing its second DiT. + # Ideogram 4 needs two DiTs assembled per-component, so there's no transformer-only + # single-file or GGUF load: the explicit kinds must be rejected up front (before a load evicts + # a working model), not assembled into a pipeline missing its second DiT. backend = DiffusionBackend() (tmp_path / "model.gguf").write_bytes(b"x") with pytest.raises(ValueError, match = "full diffusers pipeline"): @@ -1639,10 +1638,9 @@ def test_ideogram_rejects_single_file_and_gguf_kinds(fake_runtime, tmp_path): def test_generate_ideogram_defaults_keep_recommended_schedule(fake_runtime, tmp_path): - # Ideogram 4's pipeline defaults to its recommended tapered guidance_schedule - # (45x7.0 + 3x3.0, valid only at 48 steps) and REJECTS guidance_scale while the - # schedule is set. At the family's advertised defaults the backend must drop the - # constant so the recommended taper engages. + # Ideogram 4's pipeline defaults to its recommended tapered guidance_schedule (45x7.0 + 3x3.0, + # valid only at 48 steps) and REJECTS guidance_scale while the schedule is set. At the family's + # advertised defaults the backend must drop the constant so the recommended taper engages. backend = DiffusionBackend() _load_ideogram(backend, tmp_path) backend.generate(prompt = "a sloth", steps = 48, guidance = 7.0) @@ -1652,9 +1650,9 @@ def test_generate_ideogram_defaults_keep_recommended_schedule(fake_runtime, tmp_ def test_generate_ideogram_custom_guidance_nulls_schedule(fake_runtime, tmp_path): - # Any non-default request must broadcast the constant legally: guidance_scale set - # AND guidance_schedule explicitly nulled (the pipeline raises when both are set, - # and its default schedule is non-None). + # Any non-default request must broadcast the constant legally: guidance_scale set AND + # guidance_schedule explicitly nulled (the pipeline raises when both are set, and its default + # schedule is non-None). backend = DiffusionBackend() _load_ideogram(backend, tmp_path) backend.generate(prompt = "a sloth", steps = 20, guidance = 5.0) @@ -1682,9 +1680,9 @@ def test_begin_load_rejects_concurrent(monkeypatch): def test_unload_cancels_in_flight_load(fake_runtime): - # An unload (or an arbiter eviction, which calls unload) while a load's worker - # is still resolving/downloading must cancel it: load_pipeline sees the bumped - # token and aborts, so the evicted load never resurrects a pipeline into VRAM. + # An unload (or arbiter eviction, which calls unload) while a load's worker is still + # resolving/downloading must cancel it: load_pipeline sees the bumped token and aborts, so the + # evicted load never resurrects a pipeline into VRAM. backend = DiffusionBackend() fam = detect_family("unsloth/Z-Image-Turbo-GGUF") token = 7 @@ -1701,10 +1699,10 @@ def test_unload_cancels_in_flight_load(fake_runtime): def test_superseded_load_does_not_cancel_live_generation(fake_runtime): - # A superseded background load (its token was bumped by a newer load/unload) that - # finally reaches load_pipeline must bail WITHOUT signalling the current model's - # in-flight generation: the token check has to run before the cancel is set, or a - # stale worker aborts an unrelated, still-live denoise. + # A superseded background load (token bumped by a newer load/unload) that finally reaches + # load_pipeline must bail WITHOUT signalling the current model's in-flight generation: the + # token check must run before the cancel is set, or a stale worker aborts an unrelated, + # still-live denoise. import threading as _threading backend = DiffusionBackend() @@ -1904,10 +1902,10 @@ def test_generate_lock_split_keeps_status_and_unload_responsive(fake_runtime): cancel_ref = backend._active_generate_cancel assert cancel_ref is not None - # unload() signals THIS generation's cancel event, then waits for the denoise to - # actually exit before returning: callers treat its return as "VRAM is free" (the - # GPU arbiter hands the GPU to chat on it). Release the pipe once the cancel - # lands, standing in for the step callback of a real pipeline. + # unload() signals THIS generation's cancel event, then waits for the denoise to exit before + # returning: callers treat its return as "VRAM is free" (the GPU arbiter hands the GPU to chat + # on it). Release the pipe once the cancel lands, standing in for a real pipeline's step + # callback. releaser = threading.Thread(target = lambda: (cancel_ref.wait(5), release.set())) releaser.start() backend.unload() @@ -2007,9 +2005,9 @@ def test_validate_load_request(tmp_path): backend.validate_load_request("some-org/Z-Image", gguf_filename = "model.safetensors") with pytest.raises(ValueError, match = "family"): backend.validate_load_request("meta/Llama-3", gguf_filename = "q.gguf") - # A family-looking repo paired with a non-GGUF single-file name is rejected here, - # BEFORE the route evicts chat and hands over the GPU (the background load would - # otherwise be the first to notice README.md is not a checkpoint). + # A family-looking repo paired with a non-GGUF single-file name is rejected here, BEFORE the + # route evicts chat and hands over the GPU (else the background load would be the first to + # notice README.md is not a checkpoint). with pytest.raises(ValueError, match = r"\.gguf"): backend.validate_load_request("unsloth/Z-Image-Turbo-GGUF", gguf_filename = "README.md") assert ( @@ -2026,9 +2024,9 @@ def test_validate_load_request(tmp_path): backend.validate_load_request( "unsloth/Qwen-Image-2512-FP8", gguf_filename = "q.gguf", model_kind = "single_file" ) - # A remote "*-GGUF" repo loaded as a full pipeline (no single-file name) is a single-file - # GGUF repo, so from_pretrained would find no pipeline manifest and fail after chat is - # already evicted; reject it here before the GPU handoff. + # A remote "*-GGUF" repo loaded as a full pipeline (no single-file name) is a single-file GGUF + # repo, so from_pretrained finds no pipeline manifest and fails after chat is evicted; reject + # it here before the GPU handoff. with pytest.raises(ValueError, match = "GGUF"): backend.validate_load_request("unsloth/Z-Image-Turbo-GGUF", model_kind = "pipeline") # A local path with a missing child fails here (before any GPU/network work). @@ -2054,9 +2052,9 @@ def test_validate_load_request(tmp_path): def test_replacement_load_waits_for_inflight_generation(fake_runtime, tmp_path): - # A superseding load must signal the in-flight generation's cancel AND wait for - # it to release _generate_lock before allocating, so two pipelines never sit in - # VRAM at once (unlike unload(), which returns promptly without waiting). + # A superseding load must signal the in-flight generation's cancel AND wait for it to release + # _generate_lock before allocating, so two pipelines never sit in VRAM at once (unlike + # unload(), which returns promptly without waiting). import threading backend = DiffusionBackend() @@ -2140,10 +2138,9 @@ def _force_cuda_target(backend, monkeypatch): def test_load_memory_mode_balanced_streams_or_falls_back(fake_runtime, tmp_path, monkeypatch): - # balanced requests streamed block-level (group) offload. Under the stub there is - # no real diffusers.hooks, so group can't engage and the applier falls back to - # whole-module offload, reporting the policy actually engaged (the real "group" - # path is GPU-verified in the bench). + # balanced requests streamed block-level (group) offload. Under the stub there's no real + # diffusers.hooks, so group can't engage and the applier falls back to whole-module offload, + # reporting the policy actually engaged (the real "group" path is GPU-verified in the bench). (tmp_path / "m.gguf").write_bytes(b"x") backend = DiffusionBackend() _force_cuda_target(backend, monkeypatch) @@ -2184,9 +2181,9 @@ def test_load_explicit_cpu_offload_engages_model_offload_on_cuda( def test_load_speed_mode_gguf_auto_defaults_and_explicit(fake_runtime, tmp_path): - # No speed_mode on a GGUF model -> auto `default` (near-lossless, compile sits - # below the quant noise floor). compile itself only engages on CUDA, so on this - # CPU stub no optim need engage, but the resolved mode is `default`. + # No speed_mode on a GGUF model -> auto `default` (near-lossless, compile sits below the quant + # noise floor). compile only engages on CUDA, so on this CPU stub no optim engages, but the + # resolved mode is `default`. (tmp_path / "m.gguf").write_bytes(b"x") backend = DiffusionBackend() status = backend.load_pipeline(str(tmp_path), gguf_filename = "m.gguf", family_override = "z-image") @@ -2304,8 +2301,8 @@ def test_explicit_off_load_skips_dense_quant_path(fake_runtime, tmp_path, monkey def test_speed_off_load_suppresses_auto_dtype_quant(fake_runtime, tmp_path, monkeypatch): # An explicit Speed="off" (bit-exact) load with an UNSET dtype must stay GGUF-as-is: the auto - # dtype default must NOT promote it to a quantized + compiled build, which would silently break - # the user's bit-exact request. The dense gate must never be consulted. + # dtype default must NOT promote it to a quantized + compiled build (silently breaking the + # bit-exact request). The dense gate must never be consulted. from core.inference import diffusion as dmod monkeypatch.setattr( @@ -2454,9 +2451,9 @@ def test_transformer_quant_falls_back_to_gguf_on_failure(fake_runtime, tmp_path, def test_transformer_quant_skipped_when_plan_offloads(fake_runtime, tmp_path, monkeypatch): - # The dense bf16 transformer only fits resident, so when the memory plan would - # offload (here low_vram) the fast path is skipped and GGUF loads instead -- the - # dense transformer is never even loaded. + # The dense bf16 transformer only fits resident, so when the memory plan would offload (here + # low_vram) the fast path is skipped and GGUF loads instead -- the dense transformer is never + # loaded. from core.inference import diffusion as dmod backend = DiffusionBackend() @@ -2485,9 +2482,8 @@ def test_dense_quant_skipped_when_dense_transformer_does_not_fit( fake_runtime, tmp_path, monkeypatch ): # The GGUF fits resident (plan `none`), but the DENSE bf16 transformer the fast path - # materializes does not. The fast path must be skipped up front (preflighted against - # the dense transformer, not the GGUF), and GGUF loads RESIDENT -- not evicted, OOMed - # in finalization, then offloaded. + # materializes does not. The fast path must be skipped up front (preflighted against the dense + # transformer, not the GGUF), and GGUF loads RESIDENT -- not evicted, OOMed, then offloaded. from core.inference import diffusion as dmod backend = DiffusionBackend() @@ -2537,9 +2533,9 @@ def test_dense_quant_skipped_when_dense_transformer_does_not_fit( def test_dense_quant_prequant_skips_dense_refit(fake_runtime, tmp_path, monkeypatch): - # With a prequant checkpoint, the fast path loads the small quantized file, not the - # dense bf16 -- so the dense-transformer re-check must NOT run and must NOT decline the - # fast path, even when the base's dense shards happen to be cached and large. + # With a prequant checkpoint, the fast path loads the small quantized file, not the dense + # bf16 -- so the dense-transformer re-check must NOT run and must NOT decline the fast path, + # even when the base's dense shards are cached and large. from core.inference import diffusion as dmod backend = DiffusionBackend() @@ -2590,11 +2586,10 @@ def test_dense_quant_prequant_skips_dense_refit(fake_runtime, tmp_path, monkeypa def test_dense_quant_unusable_prequant_path_runs_dense_refit(fake_runtime, tmp_path, monkeypatch): - # A request-supplied transformer_prequant_path the loader will refuse (missing, or - # outside UNSLOTH_ALLOW_LOCAL_PREQUANT_PATH) resolves to NO usable prequant source, - # so the dense-transformer fit re-check MUST run: with real device budgets it is what - # declines the fast path up front instead of evicting the resident pipeline and - # OOMing in the dense bf16 fallback _load_dense_quant_pipeline falls into. + # A request-supplied transformer_prequant_path the loader refuses (missing, or outside + # UNSLOTH_ALLOW_LOCAL_PREQUANT_PATH) resolves to NO usable prequant source, so the + # dense-transformer fit re-check MUST run: with real device budgets it declines the fast path + # up front instead of evicting the resident pipeline and OOMing in the dense bf16 fallback. from core.inference import diffusion as dmod backend = DiffusionBackend() @@ -2644,10 +2639,10 @@ def test_dense_quant_unusable_prequant_path_runs_dense_refit(fake_runtime, tmp_p def test_transformer_quant_unsupported_scheme_skips_dense_download( fake_runtime, tmp_path, monkeypatch ): - # An explicit unsupported scheme (select_transformer_quant_scheme -> None) must fail - # the dense path BEFORE materialising the multi-GB dense transformer, then fall back - # to GGUF -- otherwise the download runs under the load lock during finalization - # after the old model was already evicted, only to fail at quantize. + # An explicit unsupported scheme (select_transformer_quant_scheme -> None) must fail the dense + # path BEFORE materialising the multi-GB dense transformer, then fall back to GGUF -- else the + # download runs under the load lock during finalization after the old model was evicted, only + # to fail at quantize. from core.inference import diffusion as dmod backend = DiffusionBackend() @@ -2693,10 +2688,10 @@ def test_base_file_downloaded_include_transformer_flag(): def test_dense_quant_prefetch_needed_gates(fake_runtime, monkeypatch): - # The transformer/ prefetch widens exactly when load_pipeline would take the dense-quant path: - # it defers to resolve_dense_quant_candidate (quant requested + device supported + scheme - # resolvable + no prequant checkpoint + the cache volume has disk for the extra bf16 shards). - # An explicit Speed="off" (bit-exact) load never widens. + # The transformer/ prefetch widens exactly when load_pipeline takes the dense-quant path: it + # defers to resolve_dense_quant_candidate (quant requested + device supported + scheme + # resolvable + no prequant checkpoint + disk for the extra bf16 shards). An explicit + # Speed="off" (bit-exact) load never widens. from core.inference import diffusion as dmod backend = DiffusionBackend() @@ -2729,9 +2724,9 @@ def test_dense_quant_prefetch_needed_gates(fake_runtime, monkeypatch): assert seen[-1] == "auto" # A definite-offload memory policy forces load_pipeline onto offload regardless of the dense # candidate's smaller footprint, so the dense build never runs and the widened prefetch would - # download base transformer/ shards the offloaded GGUF path never uses (and a disk-full there - # has no GGUF fallback). balanced / low_vram (and the legacy cpu_offload flag when no - # memory_mode overrides it) must NOT widen, even though the candidate itself is dense-viable. + # download base transformer/ shards the offloaded GGUF path never uses (disk-full there has no + # GGUF fallback). balanced / low_vram (and the legacy cpu_offload flag absent a memory_mode) + # must NOT widen, even though the candidate is dense-viable. before = len(seen) assert ( backend._dense_quant_prefetch_needed( @@ -2773,41 +2768,39 @@ def test_dense_quant_prefetch_needed_gates(fake_runtime, monkeypatch): backend._dense_quant_prefetch_needed(fam, {"transformer_quant": "fp8", "speed_mode": "off"}) is False ) - # A PREQUANT candidate loads the small pre-quantized checkpoint (+ config / companions), - # NOT the base repo's dense transformer/ shards, so the widened prefetch must NOT fire -- - # otherwise it defeats the prequant download savings and can hard-fail begin_load (no GGUF - # fallback there) on a disk-full. + # A PREQUANT candidate loads the small pre-quantized checkpoint (+ config / companions), NOT + # the base repo's dense transformer/ shards, so the widened prefetch must NOT fire -- else it + # defeats the prequant savings and can hard-fail begin_load (no GGUF fallback) on a disk-full. monkeypatch.setattr( dmod, "resolve_dense_quant_candidate", lambda **kw: types.SimpleNamespace(prequant = True) ) assert backend._dense_quant_prefetch_needed(fam, {"transformer_quant": "fp8"}) is False - # No viable candidate at all (unsupported scheme / no disk room) -> never widen. The disk - # guard here is exactly what averts filling the cache volume and hard-failing the load - # instead of falling back to the GGUF. + # No viable candidate (unsupported scheme / no disk room) -> never widen. The disk guard here + # averts filling the cache volume and hard-failing the load instead of falling back to GGUF. monkeypatch.setattr(dmod, "resolve_dense_quant_candidate", lambda **kw: None) assert backend._dense_quant_prefetch_needed(fam, {"transformer_quant": "fp8"}) is False def test_diffusion_status_response_carries_resolved(): # The backend records per-control auto-policy provenance (build_resolved_record) on - # state.resolved; the response model must DECLARE the field or Pydantic's default - # extra='ignore' silently drops it, leaving that plumbing dead (never reaching a client). + # state.resolved; the response model must DECLARE the field or Pydantic's extra='ignore' drops + # it, leaving that plumbing dead (never reaching a client). from models.inference import DiffusionStatusResponse rec = {"transformer_quant": {"value": "fp8", "source": "auto", "reason": "blackwell"}} resp = DiffusionStatusResponse(loaded = True, resolved = rec) # The typed field coerces the plain record into DiffusionResolvedControl objects; the # serialized form must round-trip back to the record, proving the field is DECLARED and not - # silently dropped by Pydantic's default extra='ignore'. + # dropped by Pydantic's extra='ignore'. assert resp.model_dump()["resolved"] == rec # Absent by default (nothing resolved / native engine). assert DiffusionStatusResponse(loaded = False).resolved is None def test_companion_cache_bytes_local_dir_excludes_transformer(tmp_path): - # A LOCAL diffusers base: sum the on-disk VAE / text-encoder weights so auto memory - # planning sees the resident companions, but exclude transformer/ (the GGUF supplies - # it) and non-weight files. A folded-to-zero companion could OOM a resident plan. + # A LOCAL diffusers base: sum the on-disk VAE / text-encoder weights so auto memory planning + # sees the resident companions, but exclude transformer/ (the GGUF supplies it) and non-weight + # files. A folded-to-zero companion could OOM a resident plan. (tmp_path / "vae").mkdir() (tmp_path / "vae" / "diffusion_pytorch_model.safetensors").write_bytes(b"x" * 100) (tmp_path / "text_encoder").mkdir() @@ -2820,12 +2813,11 @@ def test_companion_cache_bytes_local_dir_excludes_transformer(tmp_path): def test_plan_memory_dense_replan_does_not_double_count_prefetched_transformer(monkeypatch): - # Re-planning the dense transformer-quant candidate: the dense path prefetches the - # base repo's transformer/ shards into the SAME blob cache _companion_cache_bytes - # sums. If the re-plan read that cache it would count the transformer TWICE (once as - # transformer_resident_override_mib, once as a "companion") and force offload even - # when the quantised artifact fits resident. The re-plan must use the auto-policy's - # companion estimate instead. Here the cache is stubbed to the transformer-inflated + # Re-planning the dense transformer-quant candidate: the dense path prefetches the base repo's + # transformer/ shards into the SAME blob cache _companion_cache_bytes sums. If the re-plan read + # that cache it would count the transformer TWICE (as transformer_resident_override_mib and as + # a "companion") and force offload even when the quantised artifact fits resident. The re-plan + # must use the auto-policy's companion estimate. Here the cache is stubbed to the inflated # value; the plan must still stay resident. from core.inference import diffusion as dmod from core.inference.diffusion_memory import OFFLOAD_NONE, DeviceMemory @@ -2867,8 +2859,8 @@ def test_plan_memory_dense_replan_does_not_double_count_prefetched_transformer(m def test_reset_step_cache_helper_is_best_effort(): # Prefers the real diffusers CacheMixin hook (_reset_stateful_cache): on a genuine Flux / # QwenImage transformer that is the reset entry point, and reset_stateful_hooks lives only on - # the HookRegistry (getattr for it on the transformer returns None), so the old lookup was a - # silent no-op that left stale FBCache residuals for the next generation. + # the HookRegistry (getattr on the transformer returns None), so the old lookup was a silent + # no-op that left stale FBCache residuals for the next generation. calls = [] pipe = types.SimpleNamespace( transformer = types.SimpleNamespace(_reset_stateful_cache = lambda: calls.append("real")) @@ -2942,9 +2934,8 @@ def test_prefetch_returns_snapshot_dir_for_manifest(monkeypatch): def test_pipeline_load_uses_predownloaded_dir(fake_runtime, tmp_path): - # With a prefetched snapshot, from_pretrained must receive the local dir -- - # its own hub sweep would re-download the root packaged singles the scoped - # prefetch skips (24 GB per FLUX.1 repo). + # With a prefetched snapshot, from_pretrained must receive the local dir -- its own hub sweep + # would re-download the root packaged singles the scoped prefetch skips (24 GB per FLUX.1 repo). backend = DiffusionBackend() backend.load_pipeline( "unsloth/Qwen-Image-2512-bnb-4bit", diff --git a/studio/backend/tests/test_video_backend.py b/studio/backend/tests/test_video_backend.py index 8d8e9824fe..828b7b3eec 100644 --- a/studio/backend/tests/test_video_backend.py +++ b/studio/backend/tests/test_video_backend.py @@ -121,11 +121,11 @@ class _FakeTransformer: return object() -# ── Wan2.2 fakes: a per-DiT trackable transformer so the dual-DiT optimisation -# tests can assert speed / cache / attention engaged on BOTH experts, plus two -# pipeline fakes -- single-DiT (TI2V-5B) and dual-DiT MoE (A14B). The MoE __call__ -# carries guidance_scale_2 so the cfg2 signature-gate actually exercises; the -# single-DiT __call__ omits it so the gate proves it is NOT threaded there. +# ── Wan2.2 fakes: a per-DiT trackable transformer so the dual-DiT optimisation tests can assert +# speed / cache / attention engaged on BOTH experts, plus two pipeline fakes -- single-DiT +# (TI2V-5B) and dual-DiT MoE (A14B). The MoE __call__ carries guidance_scale_2 so the cfg2 +# signature-gate exercises; the single-DiT __call__ omits it so the gate proves it is NOT +# threaded there. class _FakeWanDiT: @@ -291,11 +291,10 @@ class _FakeWanPipelineSingle: return _FakeWanPipeMoE() if moe else _FakeWanPipeSingle() -# ── HunyuanVideo-1.5 fakes: the __call__ signature has NO guidance kwarg and NO -# callback_on_step_end (matching pipeline_hunyuan_video1_5.py in diffusers 0.39), -# a guider object carries the CFG scale, and the denoise loop drives -# scheduler.step -- so the guider write and the scheduler-wrap progress/cancel -# paths actually exercise. +# ── HunyuanVideo-1.5 fakes: __call__ has NO guidance kwarg and NO callback_on_step_end +# (matching pipeline_hunyuan_video1_5.py in diffusers 0.39), a guider object carries the CFG +# scale, and the denoise loop drives scheduler.step -- so the guider write and the scheduler-wrap +# progress/cancel paths exercise. class _FakeHV15Scheduler: @@ -477,8 +476,8 @@ def test_validate_gates_base_repo_and_local_paths(tmp_path): def test_validate_rejects_kind_extension_mismatch(tmp_path): backend = VideoBackend() # model_kind single_file with a .gguf file, or gguf with a non-.gguf file, must be rejected - # BEFORE the GPU handoff (mirrors the image loader), instead of failing in the wrong - # single-file loader after the route evicted the resident model. + # BEFORE the GPU handoff (mirrors the image loader), not fail in the wrong single-file loader + # after the route evicted the resident model. with pytest.raises(ValueError, match = "needs model_kind 'gguf'"): backend.validate_load_request( "unsloth/LTX-2.3-GGUF", @@ -506,9 +505,9 @@ def test_validate_rejects_local_file_suffix_kind_mismatch(tmp_path): backend = VideoBackend() # A local FILE is handed straight to the gguf/single_file loader: _resolve_checkpoint_path # returns the file itself, IGNORING gguf_filename, so the file's OWN suffix must match the - # kind. A .gguf file picked as single_file (or a .safetensors file picked as gguf) slips - # past the gguf_filename suffix checks, so it must be rejected HERE, before the route evicts - # the resident GPU owner and then fails deep in from_single_file / the GGUF reader. + # kind. A .gguf picked as single_file (or a .safetensors picked as gguf) slips past the + # gguf_filename suffix checks, so reject it HERE, before the route evicts the resident GPU + # owner and fails deep in from_single_file / the GGUF reader. gguf_file = tmp_path / "ltx.gguf" gguf_file.write_bytes(b"weights") safetensors_file = tmp_path / "ltx.safetensors" @@ -551,8 +550,8 @@ def test_validate_rejects_local_file_suffix_kind_mismatch(tmp_path): def test_validate_rejects_windows_shaped_missing_checkpoint(tmp_path): backend = VideoBackend() # A missing Windows-shaped local pick (backslash path, or a C:/ drive path) must fail HERE, - # not be treated as a Hub repo and only fail after the route evicts the resident GPU owner. - # Mirrors the image loader's is_absolute()/backslash path-shaped check. + # not be treated as a Hub repo and fail after the route evicts the resident GPU owner. Mirrors + # the image loader's is_absolute()/backslash path-shaped check. with pytest.raises(ValueError, match = "does not exist"): backend.validate_load_request( "C:\\models\\ltx.gguf", @@ -587,9 +586,9 @@ def test_validate_rejects_local_pipeline_without_model_index(tmp_path): def test_validate_rejects_local_file_picked_as_pipeline(tmp_path): backend = VideoBackend() # A local FILE (a bare .safetensors) sent as a pipeline is not a diffusers directory, so - # from_pretrained would only fail deep in the background load AFTER the route evicts the - # resident GPU model. The preflight must reject it HERE -- the check gates on .exists() - # (not .is_dir()), mirroring the image loader, so it catches files as well as directories. + # from_pretrained fails deep in the background load AFTER the route evicts the resident model. + # The preflight rejects it HERE -- gating on .exists() (not .is_dir()), mirroring the image + # loader, so it catches files as well as directories. f = tmp_path / "ltx-2.safetensors" f.write_bytes(b"x") with pytest.raises(ValueError, match = "model_index.json"): @@ -599,9 +598,9 @@ def test_validate_rejects_local_file_picked_as_pipeline(tmp_path): def test_validate_rejects_local_base_repo_without_model_index(tmp_path): backend = VideoBackend() # A local base_repo dir that is NOT a diffusers pipeline (no model_index.json) passes the - # any-existing-path trust check, but the base loads via from_pretrained (needs model_index), - # so reject it HERE before the route hands the GPU to VIDEO -- the pipeline-kind shape check - # covers only repo_id, and an explicit base_repo is only meaningful for a gguf/single_file load. + # any-existing-path trust check, but the base loads via from_pretrained (needs model_index), so + # reject it HERE before the route hands the GPU to VIDEO -- the pipeline-kind shape check covers + # only repo_id, and an explicit base_repo is only meaningful for a gguf/single_file load. bad_base = tmp_path / "bare-base" bad_base.mkdir() with pytest.raises(ValueError, match = "model_index.json"): @@ -706,9 +705,9 @@ def test_detect_load_family_cached_hub_arch_fallback(monkeypatch): def test_loading_repo_ids_guards_in_flight_delete(): - # During a background load status()["loaded"] is still False, but the target repo (+ its - # companion base) is being downloaded, so the delete-cached guard needs loading_repo_ids to - # refuse deletion and avoid yanking blobs from under the in-flight download/assembly. + # During a background load status()["loaded"] is still False, but the target repo (+ companion + # base) is downloading, so the delete-cached guard needs loading_repo_ids to refuse deletion + # and avoid yanking blobs from under the in-flight download/assembly. from core.inference.video import _VideoLoadingState backend = VideoBackend() @@ -759,10 +758,10 @@ def test_load_generate_unload_gguf(fake_runtime, tmp_path): def test_load_holds_generate_lock_across_placement(fake_runtime, tmp_path, monkeypatch): # The video load must hold _generate_lock across GPU placement (apply_memory_plan) so an - # unload / arbiter eviction -- which barriers on _generate_lock before freeing -- cannot hand + # unload / arbiter eviction -- which barriers on _generate_lock before freeing -- can't hand # the GPU to another backend while a multi-GB pipeline is still being moved onto it (mirrors - # the image backend, which places + commits under this lock). Verify unload() blocks until - # placement releases the lock, and the superseded load then aborts without committing. + # the image backend). Verify unload() blocks until placement releases the lock, and the + # superseded load then aborts without committing. import threading from core.inference import video as video_mod @@ -814,11 +813,10 @@ def test_load_holds_generate_lock_across_placement(fake_runtime, tmp_path, monke def test_load_records_engaged_speed_optims(fake_runtime, tmp_path, monkeypatch): - # Regression: the load tail once re-ran the already-filtered speed_optims - # tuple through ``.items()`` as if it were still the raw applied dict, so - # every real-GPU load (where at least channels_last engages) crashed with - # 'tuple' object has no attribute 'items'. Fake runtime forces every optim - # False, so this only reproduces when one is made to engage. + # Regression: the load tail once re-ran the already-filtered speed_optims tuple through + # ``.items()`` as if it were still the raw applied dict, so every real-GPU load (where at least + # channels_last engages) crashed with 'tuple' object has no attribute 'items'. Fake runtime + # forces every optim False, so this only reproduces when one is made to engage. from core.inference import video as video_mod monkeypatch.setattr( @@ -849,12 +847,11 @@ def test_generate_defaults_from_variant(fake_runtime, tmp_path): def test_generate_resets_step_cache_only_when_engaged(fake_runtime, tmp_path): - # FBCache residuals live on the long-lived DiT(s) and survive a generation, so - # the next clip at a new resolution would crash on stale state. generate must - # reset them when a cache is engaged (diffusers 0.39 exposes - # _reset_stateful_cache on the transformer; reset_stateful_hooks only exists on - # the HookRegistry) and must not touch an uncached load. transformer_2 (the Wan - # dual expert) resets too when present. + # FBCache residuals live on the long-lived DiT(s) and survive a generation, so the next clip + # at a new resolution would crash on stale state. generate must reset them when a cache is + # engaged (diffusers 0.39 exposes _reset_stateful_cache on the transformer; reset_stateful_hooks + # only on the HookRegistry) and must not touch an uncached load. transformer_2 (the Wan dual + # expert) resets too when present. import dataclasses (tmp_path / "ltx-2.3-22b-distilled-1.1-Q4_K_M.gguf").write_bytes(b"w") @@ -882,10 +879,9 @@ def test_generate_resets_step_cache_only_when_engaged(fake_runtime, tmp_path): def test_is_ltx23_checkpoint_gguf(monkeypatch, tmp_path): - # diffusers maps every LTX-2 single file to the 2.0 config; a 2.3 checkpoint - # (9-row modulation tables in the header) must be detected so the loader - # routes to the full 2.3 assembly. A 2.0 header must not, and an unreadable - # header must fall back to the stock path (False), never raise. + # diffusers maps every LTX-2 single file to the 2.0 config; a 2.3 checkpoint (9-row modulation + # tables in the header) must be detected so the loader routes to the full 2.3 assembly. A 2.0 + # header must not, and an unreadable header falls back to the stock path (False), never raise. from core.inference.video_ltx2 import is_ltx23_checkpoint def _reader_for(shapes): @@ -986,9 +982,8 @@ def test_ltx23_split_and_variant(tmp_path): def test_ltx23_scaled_fp8_refused(monkeypatch, tmp_path): - # The Lightricks fp8 files carry .weight_scale/.input_scale companions; a - # plain dtype cast would silently corrupt them, so the loader must refuse - # with a pointer to the supported GGUF path. + # The Lightricks fp8 files carry .weight_scale/.input_scale companions; a plain dtype cast + # would corrupt them, so the loader must refuse with a pointer to the supported GGUF path. from core.inference import video_ltx2 # Stub the module tree so this also runs under the CI sim, which blocks the @@ -1068,16 +1063,16 @@ def test_hv15_cancel_unwinds_scheduler_loop(fake_runtime): assert pipe.scheduler.calls == 1 # The wrapper must restore scheduler.step even on the exception path. assert pipe.scheduler.step.__func__ is _FakeHV15Scheduler.step - # The exception unwound pipe.__call__ before its own end-of-call cleanup, so - # generate() must have freed the offload hooks itself (VRAM would otherwise - # stay onloaded until the next request). + # The exception unwound pipe.__call__ before its own end-of-call cleanup, so generate() must + # have freed the offload hooks itself (VRAM would otherwise stay onloaded until the next + # request). assert pipe.hooks_freed == 1 def test_cancel_during_export_discards_clip(fake_runtime, monkeypatch): - # A cancel that lands during the (blocking, uncancellable) export/mux must still discard - # the clip: cancel_generate() already reported success for it, so generate() must raise - # the cancelled sentinel rather than return the clip to be persisted to the gallery. + # A cancel landing during the (blocking, uncancellable) export/mux must still discard the + # clip: cancel_generate() already reported success, so generate() must raise the cancelled + # sentinel rather than return the clip to be persisted to the gallery. backend = VideoBackend() backend.load_pipeline( "hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-480p_t2v", @@ -1116,9 +1111,9 @@ def test_load_wan_ti2v_5b_pipeline(fake_runtime): def test_video_dense_speed_defaults_to_compile_profile(fake_runtime): - # A clip denoise amortises the one-time compile within a single run, so an - # UNSET speed on a dense (pipeline) load resolves to `default` -- never `max`, - # never `off`. Explicit "off" is still honored verbatim. + # A clip denoise amortises the one-time compile within a single run, so an UNSET speed on a + # dense (pipeline) load resolves to `default` -- never `max`, never `off`. Explicit "off" is + # still honored verbatim. backend = VideoBackend() status = backend.load_pipeline("Wan-AI/Wan2.2-TI2V-5B-Diffusers", model_kind = "pipeline") assert status["speed_mode"] == "default" @@ -1133,9 +1128,9 @@ def test_video_dense_speed_defaults_to_compile_profile(fake_runtime): def test_video_speed_off_suppresses_auto_dtype_quant(fake_runtime, monkeypatch): # An explicit Speed="off" (bit-exact) pipeline load with Precision left at auto must NOT - # promote the unset precision to auto-quant: doing so would engage torchao quantization (and - # then force the speed back to default), silently breaking the bit-exact request. Mirrors the - # image backend. On a dense-capable GPU (stubbed) quantize_transformer must not run. + # promote the unset precision to auto-quant: that would engage torchao quantization (and force + # speed back to default), breaking the bit-exact request. Mirrors the image backend. On a + # dense-capable GPU (stubbed) quantize_transformer must not run. import core.inference.video as video_mod monkeypatch.setattr(video_mod, "dense_transformer_supported", lambda target: True) @@ -1161,9 +1156,9 @@ def test_video_speed_off_suppresses_auto_dtype_quant(fake_runtime, monkeypatch): def test_video_step_cache_auto_from_default_schedule(fake_runtime, tmp_path): - # Unset step cache is AUTO, decided from the model's default schedule: Wan's - # 50-step default engages FBCache at load; the LTX distilled 8-step default - # keeps it off. Both are re-checked per generation (toggle test below). + # Unset step cache is AUTO, decided from the model's default schedule: Wan's 50-step default + # engages FBCache at load; the LTX distilled 8-step default keeps it off. Both are re-checked + # per generation (toggle test below). backend = VideoBackend() status = backend.load_pipeline("Wan-AI/Wan2.2-TI2V-5B-Diffusers", model_kind = "pipeline") assert status["transformer_cache"] == "fbcache" @@ -1183,9 +1178,8 @@ def test_video_step_cache_auto_from_default_schedule(fake_runtime, tmp_path): def test_video_step_cache_auto_toggles_on_actual_steps(fake_runtime): - # The AUTO decision follows the ACTUAL step count of each generation: a - # few-step request drops the load-time cache, a many-step request restores - # it. An explicit "off" never toggles. + # The AUTO decision follows the ACTUAL step count of each generation: a few-step request drops + # the load-time cache, a many-step request restores it. An explicit "off" never toggles. backend = VideoBackend() backend.load_pipeline("Wan-AI/Wan2.2-TI2V-5B-Diffusers", model_kind = "pipeline") assert backend.status()["transformer_cache"] == "fbcache" @@ -1227,9 +1221,9 @@ def test_wan_ti2v_defaults_applied(fake_runtime): def test_wan_ti2v_does_not_thread_cfg2(fake_runtime): - # The single-DiT TI2V pipeline has no guidance_scale_2 in its signature, so a - # request value must NOT be threaded (WanPipeline raises on it when boundary_ratio - # is None), even if the caller passes guidance_2. + # The single-DiT TI2V pipeline has no guidance_scale_2 in its signature, so a request value + # must NOT be threaded (WanPipeline raises on it when boundary_ratio is None), even if the + # caller passes guidance_2. backend = VideoBackend() backend.load_pipeline("Wan-AI/Wan2.2-TI2V-5B-Diffusers", model_kind = "pipeline") backend.generate(prompt = "a sloth", guidance_2 = 3.5) @@ -1278,9 +1272,9 @@ def test_wan_a14b_step_cache_applies_to_both_dits(fake_runtime): def test_wan_a14b_attention_applies_to_both_dits(fake_runtime, monkeypatch): - # An explicit attention backend must be set on both experts. The fake runtime is a - # CPU target, where the NVIDIA gate correctly drops explicit kernels; pin the gate - # open so the explicit-set path itself is what this test exercises. + # An explicit attention backend must be set on both experts. The fake runtime is a CPU target, + # where the NVIDIA gate drops explicit kernels; pin the gate open so the explicit-set path is + # what this test exercises. from core.inference import diffusion_attention as attn_mod monkeypatch.setattr(attn_mod, "_is_cuda_nvidia", lambda target: True) @@ -1311,10 +1305,9 @@ def test_wan_ti2v_single_dit_only_touches_one(fake_runtime): def test_wan_a14b_dense_quant_applies_to_both_dits(fake_runtime, monkeypatch): - # transformer_quant on a pipeline load quantises the dense DiT(s). On CPU the real - # dense path is unsupported, so stub the two quant seams to record which pipe view - # each helper saw: BOTH experts must be quantised (via the _SecondDiTView proxy), - # and status must report the engaged scheme. + # transformer_quant on a pipeline load quantises the dense DiT(s). On CPU the real dense path + # is unsupported, so stub the two quant seams to record which pipe view each helper saw: BOTH + # experts must be quantised (via the _SecondDiTView proxy), and status must report the scheme. import core.inference.video as video_mod monkeypatch.setattr(video_mod, "dense_transformer_supported", lambda target: True) @@ -1348,10 +1341,10 @@ def test_wan_a14b_dense_quant_applies_to_both_dits(fake_runtime, monkeypatch): def test_dense_quant_skipped_under_offload(fake_runtime, monkeypatch): - # Offload hooks move modules with Module.to(), which torchao quantized tensors - # reject (observed as a hard crash on the A14B gate run). When the memory plan - # resolves to any offload policy, quant must be SKIPPED, not attempted: the - # load succeeds dense and the resolved record explains why. + # Offload hooks move modules with Module.to(), which torchao quantized tensors reject + # (observed as a hard crash on the A14B gate run). When the memory plan resolves to any offload + # policy, quant must be SKIPPED, not attempted: the load succeeds dense and the record explains + # why. import core.inference.video as video_mod monkeypatch.setattr(video_mod, "dense_transformer_supported", lambda target: True) @@ -1369,9 +1362,9 @@ def test_dense_quant_skipped_under_offload(fake_runtime, monkeypatch): return "int8" monkeypatch.setattr(video_mod, "quantize_transformer", _fake_quant) - # The CPU fake target never plans an offload, so force one at the plan seam - # (frozen dataclass -> dataclasses.replace) and stub the apply step, which - # would otherwise call offload hooks the fake pipe does not have. + # The CPU fake target never plans an offload, so force one at the plan seam (frozen dataclass + # -> dataclasses.replace) and stub the apply step, which would else call offload hooks the fake + # pipe lacks. import dataclasses real_plan = video_mod.plan_diffusion_memory @@ -1399,9 +1392,9 @@ def test_dense_quant_skipped_under_offload(fake_runtime, monkeypatch): def test_wan_a14b_partial_quant_fails_the_load(fake_runtime, monkeypatch): - # If the first expert quantises but the second does not, the pipe is left at - # mismatched precision with no way back (in-place mutation), so the load must - # fail cleanly rather than run mixed with quant reported off. + # If the first expert quantises but the second doesn't, the pipe is left at mismatched + # precision with no way back (in-place mutation), so the load must fail cleanly rather than run + # mixed with quant reported off. import core.inference.video as video_mod monkeypatch.setattr(video_mod, "dense_transformer_supported", lambda target: True) @@ -1472,9 +1465,9 @@ def test_wan_validate_trusted_repos(fake_runtime): def test_wan_a14b_refuses_single_file_loads(fake_runtime): - # A single gguf/safetensors checkpoint carries only one of the A14B's two experts; - # the pipeline would pull the other dense bf16 from the base repo outside the - # memory plan, so validate refuses it up front (before any download). + # A single gguf/safetensors checkpoint carries only one of the A14B's two experts; the + # pipeline would pull the other dense bf16 from the base repo outside the memory plan, so + # validate refuses it up front (before any download). backend = VideoBackend() with pytest.raises(ValueError, match = "dual-expert"): backend.validate_load_request( @@ -1490,9 +1483,9 @@ def test_wan_a14b_refuses_single_file_loads(fake_runtime): def test_second_dit_view_write_through(): - # Attribute writes on the proxy must land on the real pipe (a helper's side - # effect would otherwise vanish with the temporary view); a ``transformer`` - # write mirrors the read property onto the second expert. + # Attribute writes on the proxy must land on the real pipe (a helper's side effect would else + # vanish with the temporary view); a ``transformer`` write mirrors the read property onto the + # second expert. from core.inference.video import _SecondDiTView pipe = types.SimpleNamespace(transformer = "t1", transformer_2 = "t2", flag = None) @@ -1649,9 +1642,9 @@ def test_predownload_base_honors_cancel_between_files(monkeypatch): def test_detect_load_family_arch_fallback_for_local_gguf(tmp_path, monkeypatch): - # A local GGUF is admitted to the Video picker by its general.architecture, but its path - # name may carry no whole-segment family token (a renamed "model.gguf"). The loader must - # resolve the same family the picker offered by reading the arch, not only the name. + # A local GGUF is admitted to the Video picker by its general.architecture, but its path name + # may carry no whole-segment family token (a renamed "model.gguf"). The loader must resolve the + # same family the picker offered by reading the arch, not only the name. from core.inference import video as vid from core.inference.video_families import detect_video_family diff --git a/studio/install_sd_cpp_prebuilt.py b/studio/install_sd_cpp_prebuilt.py index 8533fac871..a7b00861e7 100644 --- a/studio/install_sd_cpp_prebuilt.py +++ b/studio/install_sd_cpp_prebuilt.py @@ -38,17 +38,17 @@ import zipfile from pathlib import Path from typing import Optional, Sequence -# Default source: the Unsloth-built mirror, whose CPU/Apple prebuilts are compiled and -# published by unslothai/stable-diffusion.cpp (the same way unslothai/llama.cpp ships its -# prebuilts). Override with UNSLOTH_SD_CPP_REPO to point elsewhere (e.g. back to leejet). -# GPU hosts never reach here -- they run diffusers -- so only CPU/Apple assets are needed. +# Default source: the Unsloth-built mirror, whose CPU/Apple prebuilts are published by +# unslothai/stable-diffusion.cpp (like unslothai/llama.cpp ships its prebuilts). Override with +# UNSLOTH_SD_CPP_REPO to point elsewhere (e.g. back to leejet). GPU hosts never reach here (they +# run diffusers), so only CPU/Apple assets are needed. DEFAULT_REPO = "unslothai/stable-diffusion.cpp" # Upstream we fall back to if the mirror can't serve this host (mirror release missing, or # a host we don't yet build): resolve against leejet so native install still works. UPSTREAM_FALLBACK_REPO = "leejet/stable-diffusion.cpp" -# Pinned release tag for REPRODUCIBILITY: "releases/latest" silently swaps the binary -# under users on every push. Override with UNSLOTH_SD_CPP_TAG; set it empty to track -# latest. If the pinned tag is gone, install falls back to that repo's latest. +# Pinned release tag for REPRODUCIBILITY: "releases/latest" silently swaps the binary under +# users on every push. Override with UNSLOTH_SD_CPP_TAG; set it empty to track latest. If the +# pinned tag is gone, install falls back to that repo's latest. DEFAULT_TAG = "master-741-484baa4" # Back-compat alias (some callers/tests import REPO). @@ -124,9 +124,9 @@ def resolve_release_asset( sel = [a for a in pool if token in a.lower()] if sel: return sel[0] - # An EXPLICIT GPU accelerator with no matching asset is a real miss, not a CPU - # request: return None so the caller can fall back to a repo that builds it, - # rather than silently installing a CPU build for a --accelerator cuda request. + # An EXPLICIT GPU accelerator with no matching asset is a real miss, not a CPU request: + # return None so the caller can fall back to a repo that builds it, rather than installing + # a CPU build for a --accelerator cuda request. if accel in ("cuda", "vulkan", "rocm"): return None # auto / cpu -> a plain avx2 CPU build, else any windows build. @@ -293,9 +293,9 @@ def _maybe_fetch_windows_cudart(release: dict, chosen: str, target: Path) -> Non print(f"downloading CUDA runtime {cudart['name']} ...", flush = True) try: _download(cudart["browser_download_url"], dest) - # Verify integrity BEFORE extracting, like the main sd-cli archive: these DLLs are - # loaded into sd-cli.exe at runtime, so a corrupt/tampered runtime archive must be - # rejected rather than extracted next to the binary. + # Verify integrity BEFORE extracting, like the main sd-cli archive: these DLLs load into + # sd-cli.exe at runtime, so a corrupt/tampered runtime archive must be rejected rather than + # extracted next to the binary. _verify_sha256(dest, cudart.get("digest")) with zipfile.ZipFile(dest) as zf: _safe_extractall(zf, target) @@ -346,9 +346,9 @@ def _resolve_with_fallback( ``--print-asset`` so both honour the same fallback.""" tag = _pinned_tag() primary = _repo() - # Fall back to upstream ONLY when the user did not pin a repo (env unset) and the - # built-in default is in use: an explicit UNSLOTH_SD_CPP_REPO (even one equal to the - # default) gets exactly that repo, with no surprise upstream substitution. + # Fall back to upstream ONLY when the user didn't pin a repo (env unset) and the built-in + # default is in use: an explicit UNSLOTH_SD_CPP_REPO (even one equal to the default) gets + # exactly that repo, with no surprise upstream substitution. repo_pinned = bool((os.environ.get("UNSLOTH_SD_CPP_REPO") or "").strip()) allow_upstream = ( not repo_pinned and primary == DEFAULT_REPO and DEFAULT_REPO != UPSTREAM_FALLBACK_REPO @@ -375,9 +375,8 @@ def _resolve_with_fallback( ) if release is not None and chosen: if repo != primary: - # Diagnostic goes to stderr, not stdout: --print-asset documents its - # stdout as the asset name only, so a caller parsing it as a single line - # must not see this fallback log mixed in. + # Diagnostic goes to stderr, not stdout: --print-asset documents its stdout as the + # asset name only, so a caller parsing it as a single line must not see this log. print( f"falling back to {repo} for {platform.system()}/{platform.machine()}", file = sys.stderr, @@ -442,9 +441,9 @@ def install( _make_executable(sd_server) if sd_server is not None: print(f"installed sd-server -> {sd_server}", flush = True) - # Ownership marker (the same one setup.sh/_is_studio_root use, and setup.ps1 writes into - # the Node sibling dir) so the uninstaller can tell a Studio-installed sd.cpp from a user's - # own stable-diffusion.cpp checkout beside a custom Studio root, and delete only ours. + # Ownership marker (the same one setup.sh/_is_studio_root use, and setup.ps1 writes into the + # Node sibling dir) so the uninstaller can tell a Studio-installed sd.cpp from a user's own + # stable-diffusion.cpp checkout beside a custom Studio root, and delete only ours. try: (target / ".unsloth-studio-owned").touch() except OSError: @@ -464,9 +463,9 @@ def main(argv: Optional[list[str]] = None) -> int: args = p.parse_args(argv) if args.print_asset: - # Route through the same primary/fallback resolution as install(), so a host the - # mirror does not build (e.g. a Linux Vulkan request) reports the upstream asset - # it would actually download instead of a false "no matching prebuilt". + # Route through the same primary/fallback resolution as install(), so a host the mirror + # doesn't build (e.g. a Linux Vulkan request) reports the upstream asset it would actually + # download instead of a false "no matching prebuilt". _used, _release, chosen = _resolve_with_fallback(args.accelerator, None) print(chosen or "(no matching prebuilt; build from source)") return 0 if chosen else 2