Merge branch 'diffusion-phase9-prequant' into diffusion-phase10-attention

# Conflicts:
#	studio/backend/tests/test_diffusion_routes.py
This commit is contained in:
Daniel Han 2026-07-01 11:30:23 +00:00
commit b84d7ea7f3
285 changed files with 23122 additions and 2775 deletions

View file

@ -40,9 +40,13 @@ def bench_pytorch(repo, gguf, resolutions, steps, seed, iters):
backend = DiffusionBackend()
for speed in ("off", "default"):
backend.begin_load(repo, gguf_filename = gguf, speed_mode = speed)
deadline = time.time() + 1800 # 30 min: a stuck download/load must not hang forever
while backend.load_progress().get("phase") != "ready":
if backend.load_progress().get("phase") == "error":
raise RuntimeError(backend.load_progress())
prog = backend.load_progress()
if prog.get("phase") == "error":
raise RuntimeError(prog)
if time.time() > deadline:
raise TimeoutError(f"load timed out (last progress: {prog})")
time.sleep(0.5)
for res in resolutions:
@ -128,11 +132,13 @@ def main(argv = None) -> int:
)
p.add_argument(
"--vae",
default = "/mnt/disks/unslothai/ubuntu/workspace_81/sdcpp_assets/flux_vae/ae.safetensors",
default = None,
help = "VAE safetensors for sd.cpp (required when benchmarking the sd.cpp engine)",
)
p.add_argument(
"--llm",
default = "/mnt/disks/unslothai/ubuntu/workspace_81/sdcpp_assets/qwen3_te/Qwen3-4B-Instruct-2507-Q4_K_M.gguf",
default = None,
help = "text-encoder GGUF for sd.cpp (required when benchmarking the sd.cpp engine)",
)
p.add_argument("--resolutions", default = "512,1024")
p.add_argument("--steps", type = int, default = 8)

View file

@ -142,8 +142,10 @@ def _psnr(ref_png: Path, cand_png: Path) -> float:
import numpy as np
from PIL import Image
a = np.asarray(Image.open(ref_png).convert("RGB"), dtype = np.float64)
b = np.asarray(Image.open(cand_png).convert("RGB"), dtype = np.float64)
with Image.open(ref_png) as im_a:
a = np.asarray(im_a.convert("RGB"), dtype = np.float64)
with Image.open(cand_png) as im_b:
b = np.asarray(im_b.convert("RGB"), dtype = np.float64)
if a.shape != b.shape:
# Different geometry means the comparison is meaningless; report worst case.
return 0.0
@ -368,9 +370,14 @@ def _compare(args: argparse.Namespace) -> int:
print(" refusing noisy comparison (pass --force-compare to override).", flush = True)
return 2
# PSNR vs the stored reference image.
# PSNR vs the stored reference image. The baseline stores an absolute reference_png,
# which breaks if the baseline directory was copied/moved, so fall back to reference.png
# next to the baseline JSON. A still-missing reference is a failure below, not a silent
# pass -- otherwise the benchmark would report PASS having done no image comparison.
ref_png = Path(baseline.get("accuracy", {}).get("reference_png", ""))
psnr = _psnr(ref_png, args._image_out) if ref_png.exists() else float("nan")
if not ref_png.is_file():
ref_png = baseline_path.parent / "reference.png"
psnr = _psnr(ref_png, args._image_out) if ref_png.is_file() else float("nan")
base_gen = baseline.get("generate", {})
cur_gen = metrics["generate"]
@ -402,7 +409,9 @@ def _compare(args: argparse.Namespace) -> int:
)
if base_peak and cur_peak and vram_reg > args.max_vram_regression:
failures.append(f"peak VRAM +{vram_reg * 100:.1f}% > {args.max_vram_regression * 100:.0f}%")
if not math.isnan(psnr) and psnr < args.min_psnr:
if math.isnan(psnr):
failures.append("PSNR reference image missing; cannot verify output quality")
elif psnr < args.min_psnr:
failures.append(f"PSNR {psnr:.2f}dB < {args.min_psnr:.1f}dB (output changed)")
if failures:

View file

@ -186,6 +186,18 @@ def _wait_for_load(backend: Any, timeout_s: int = 3600) -> None:
def _hf_file_size_mib(repo: str, filename: str) -> Optional[int]:
# A local model dir / file: stat it directly. The Hub lookup below returns None for
# a local path, which would drop every candidate from _recommend (file_size_mib None).
try:
local = Path(repo).expanduser()
if local.is_dir():
f = local / filename
if f.is_file():
return int(f.stat().st_size // (1024 * 1024))
elif local.is_file():
return int(local.stat().st_size // (1024 * 1024))
except Exception:
pass
try:
from huggingface_hub import HfApi
info = HfApi().model_info(repo, files_metadata = True, token = os.environ.get("HF_TOKEN"))
@ -268,6 +280,11 @@ def _compare(
clip_sim.append(clip.image_similarity(img, ref))
def _mean(xs: list[float]) -> Optional[float]:
# Preserve +inf: an identical render (reference vs itself, or a lossless
# quant/offload) scores PSNR=inf, which is exactly the case this harness
# verifies; dropping it as non-finite would print "-" instead of "inf".
if xs and any(x == math.inf for x in xs):
return math.inf
finite = [x for x in xs if math.isfinite(x)]
return round(sum(finite) / len(finite), 4) if finite else None

View file

@ -16,7 +16,7 @@ import numpy as np
BASE = "Tongyi-MAI/Z-Image-Turbo"
PROMPT = "A cinematic photograph of a red fox in a snowy forest at dawn, highly detailed"
OUT = Path("/mnt/disks/unslothai/ubuntu/workspace_81/outputs/quant_research/nvfp4_images")
OUT = Path(__file__).resolve().parent.parent / "outputs" / "quant_research" / "nvfp4_images"
def _psnr(a, b):
@ -84,11 +84,15 @@ def main(argv = None) -> int:
p.add_argument("--seed", type = int, default = 42)
p.add_argument("--iters", type = int, default = 3)
p.add_argument("--min-feat", type = int, default = 512)
p.add_argument("--out-dir", default = None, help = "image output dir (default: repo outputs/)")
args = p.parse_args(argv)
steps, res, seed, mf = args.steps, args.res, args.seed, args.min_feat
import torch
import torch.nn as nn
global OUT
if args.out_dir:
OUT = Path(args.out_dir).expanduser()
OUT.mkdir(parents = True, exist_ok = True)
def filt(mod, fqn = ""):

View file

@ -151,7 +151,12 @@ def main(argv = None) -> int:
flush = True,
)
ok = (leak_psnr == float("inf")) and (def_t < off_t) and (_psnr(off_img, def_img) >= 30)
ok = (
(leak_psnr == float("inf"))
and (bal_psnr == float("inf")) # check 3: balanced must be bit-identical to off
and (def_t < off_t)
and (_psnr(off_img, def_img) >= 30)
)
print(f"\nPERF-VERIFY {'OK' if ok else 'CHECK'}", flush = True)
return 0 if ok else 1

View file

@ -26,7 +26,7 @@ import numpy as np
BASE = "Tongyi-MAI/Z-Image-Turbo"
PROMPT = "A cinematic photograph of a red fox in a snowy forest at dawn, highly detailed"
ROOT = Path("/mnt/disks/unslothai/ubuntu/workspace_81/outputs/quant_research")
ROOT = Path(__file__).resolve().parent.parent / "outputs" / "quant_research"
CKPT = ROOT / "prequant_fp8" / "transformer_fp8_state.pt"
OUT = ROOT / "prequant_images"
MIN_FEAT = 512

View file

@ -27,7 +27,7 @@ REPO = "unsloth/Z-Image-Turbo-GGUF"
GGUF = "z-image-turbo-Q4_K_M.gguf"
BASE = "Tongyi-MAI/Z-Image-Turbo"
PROMPT = "A cinematic photograph of a red fox in a snowy forest at dawn, highly detailed"
OUT = Path("/mnt/disks/unslothai/ubuntu/workspace_81/outputs/quant_research/probe_images")
OUT = Path(__file__).resolve().parent.parent / "outputs" / "quant_research" / "probe_images"
def _psnr(a, b):
@ -39,17 +39,19 @@ _LPIPS = {"fn": None}
def _lpips(ref_arr, arr):
"""Perceptual LPIPS (alexnet) vs reference; lower is closer. None if unavailable."""
"""Perceptual LPIPS (alexnet) vs reference; lower is closer. None if unavailable.
Runs on CPU so the scorer never holds CUDA memory: each row resets peak VRAM, so a
resident GPU LPIPS module would inflate the reported load/gen VRAM and could even OOM."""
try:
import torch
import lpips
if _LPIPS["fn"] is None:
_LPIPS["fn"] = lpips.LPIPS(net = "alex", verbose = False).cuda().eval()
_LPIPS["fn"] = lpips.LPIPS(net = "alex", verbose = False).eval()
def t(x):
t = torch.from_numpy(x).float().permute(2, 0, 1).unsqueeze(0) / 127.5 - 1.0
return t.cuda()
return torch.from_numpy(x).float().permute(2, 0, 1).unsqueeze(0) / 127.5 - 1.0
with torch.no_grad():
return float(_LPIPS["fn"](t(ref_arr), t(arr)).item())
@ -111,7 +113,7 @@ def _quant_config(name):
return MXDynamicActivationMXWeightConfig(
activation_dtype = torch.float8_e4m3fn, weight_dtype = torch.float8_e4m3fn
)
except TypeError:
except (TypeError, AttributeError):
return MXDynamicActivationMXWeightConfig()
raise ValueError(name)

View file

@ -1208,9 +1208,10 @@ def check_js_file(content: str, filename: str, package: str) -> list[Finding]:
HIGH,
package,
filename,
f"Python wheel ships large ({len(content) // 1024} KB) JS bundle "
"(uncommon; manually review)",
"",
# Size stays in evidence, not the check label, so the baseline key
# does not drift when a wheel's bundle grows by a few KB.
"Python wheel ships large JS bundle (uncommon; manually review)",
f"{len(content) // 1024} KB JS bundle",
)
)
return findings

View file

@ -1181,7 +1181,7 @@
{
"package": "tensorboard",
"file": "tensorboard/plugins/projector/tf_projector_plugin/projector_binary.js",
"check": "Python wheel ships large (1918 KB) JS bundle (uncommon; manually review)",
"check": "Python wheel ships large JS bundle (uncommon; manually review)",
"severity": "HIGH",
"evidence": ""
},

View file

@ -19,17 +19,20 @@ from __future__ import annotations
import argparse
import logging
import os
import sys
import time
from pathlib import Path
import numpy as np
BACKEND = Path(__file__).resolve().parent.parent / "studio" / "backend"
_REPO = Path(__file__).resolve().parent.parent
_RESEARCH = _REPO / "outputs" / "quant_research"
BACKEND = _REPO / "studio" / "backend"
BASE = "Tongyi-MAI/Z-Image-Turbo"
CKPT = "/mnt/disks/unslothai/ubuntu/workspace_81/outputs/quant_research/prequant_fp8/transformer_fp8.pt"
CKPT = os.environ.get("PREQUANT_CKPT", str(_RESEARCH / "prequant_fp8" / "transformer_fp8.pt"))
PROMPT = "A cinematic photograph of a red fox in a snowy forest at dawn, highly detailed"
OUT = Path("/mnt/disks/unslothai/ubuntu/workspace_81/outputs/quant_research/prequant_verify_images")
OUT = Path(os.environ.get("PREQUANT_OUT_DIR", str(_RESEARCH / "prequant_verify_images")))
logging.basicConfig(level = logging.INFO, format = "%(message)s")
LOGGER = logging.getLogger("verify_prequant")