Merge branch 'diffusion-phase11-consumer-int8' into diffusion-phase12-fbcache
This commit is contained in:
commit
2a5713aff5
289 changed files with 23436 additions and 2808 deletions
|
|
@ -40,9 +40,13 @@ def bench_pytorch(repo, gguf, resolutions, steps, seed, iters):
|
|||
backend = DiffusionBackend()
|
||||
for speed in ("off", "default"):
|
||||
backend.begin_load(repo, gguf_filename = gguf, speed_mode = speed)
|
||||
deadline = time.time() + 1800 # 30 min: a stuck download/load must not hang forever
|
||||
while backend.load_progress().get("phase") != "ready":
|
||||
if backend.load_progress().get("phase") == "error":
|
||||
raise RuntimeError(backend.load_progress())
|
||||
prog = backend.load_progress()
|
||||
if prog.get("phase") == "error":
|
||||
raise RuntimeError(prog)
|
||||
if time.time() > deadline:
|
||||
raise TimeoutError(f"load timed out (last progress: {prog})")
|
||||
time.sleep(0.5)
|
||||
for res in resolutions:
|
||||
|
||||
|
|
@ -128,11 +132,13 @@ def main(argv = None) -> int:
|
|||
)
|
||||
p.add_argument(
|
||||
"--vae",
|
||||
default = "/mnt/disks/unslothai/ubuntu/workspace_81/sdcpp_assets/flux_vae/ae.safetensors",
|
||||
default = None,
|
||||
help = "VAE safetensors for sd.cpp (required when benchmarking the sd.cpp engine)",
|
||||
)
|
||||
p.add_argument(
|
||||
"--llm",
|
||||
default = "/mnt/disks/unslothai/ubuntu/workspace_81/sdcpp_assets/qwen3_te/Qwen3-4B-Instruct-2507-Q4_K_M.gguf",
|
||||
default = None,
|
||||
help = "text-encoder GGUF for sd.cpp (required when benchmarking the sd.cpp engine)",
|
||||
)
|
||||
p.add_argument("--resolutions", default = "512,1024")
|
||||
p.add_argument("--steps", type = int, default = 8)
|
||||
|
|
|
|||
|
|
@ -142,8 +142,10 @@ def _psnr(ref_png: Path, cand_png: Path) -> float:
|
|||
import numpy as np
|
||||
from PIL import Image
|
||||
|
||||
a = np.asarray(Image.open(ref_png).convert("RGB"), dtype = np.float64)
|
||||
b = np.asarray(Image.open(cand_png).convert("RGB"), dtype = np.float64)
|
||||
with Image.open(ref_png) as im_a:
|
||||
a = np.asarray(im_a.convert("RGB"), dtype = np.float64)
|
||||
with Image.open(cand_png) as im_b:
|
||||
b = np.asarray(im_b.convert("RGB"), dtype = np.float64)
|
||||
if a.shape != b.shape:
|
||||
# Different geometry means the comparison is meaningless; report worst case.
|
||||
return 0.0
|
||||
|
|
@ -368,9 +370,14 @@ def _compare(args: argparse.Namespace) -> int:
|
|||
print(" refusing noisy comparison (pass --force-compare to override).", flush = True)
|
||||
return 2
|
||||
|
||||
# PSNR vs the stored reference image.
|
||||
# PSNR vs the stored reference image. The baseline stores an absolute reference_png,
|
||||
# which breaks if the baseline directory was copied/moved, so fall back to reference.png
|
||||
# next to the baseline JSON. A still-missing reference is a failure below, not a silent
|
||||
# pass -- otherwise the benchmark would report PASS having done no image comparison.
|
||||
ref_png = Path(baseline.get("accuracy", {}).get("reference_png", ""))
|
||||
psnr = _psnr(ref_png, args._image_out) if ref_png.exists() else float("nan")
|
||||
if not ref_png.is_file():
|
||||
ref_png = baseline_path.parent / "reference.png"
|
||||
psnr = _psnr(ref_png, args._image_out) if ref_png.is_file() else float("nan")
|
||||
|
||||
base_gen = baseline.get("generate", {})
|
||||
cur_gen = metrics["generate"]
|
||||
|
|
@ -402,7 +409,9 @@ def _compare(args: argparse.Namespace) -> int:
|
|||
)
|
||||
if base_peak and cur_peak and vram_reg > args.max_vram_regression:
|
||||
failures.append(f"peak VRAM +{vram_reg * 100:.1f}% > {args.max_vram_regression * 100:.0f}%")
|
||||
if not math.isnan(psnr) and psnr < args.min_psnr:
|
||||
if math.isnan(psnr):
|
||||
failures.append("PSNR reference image missing; cannot verify output quality")
|
||||
elif psnr < args.min_psnr:
|
||||
failures.append(f"PSNR {psnr:.2f}dB < {args.min_psnr:.1f}dB (output changed)")
|
||||
|
||||
if failures:
|
||||
|
|
|
|||
|
|
@ -186,6 +186,18 @@ def _wait_for_load(backend: Any, timeout_s: int = 3600) -> None:
|
|||
|
||||
|
||||
def _hf_file_size_mib(repo: str, filename: str) -> Optional[int]:
|
||||
# A local model dir / file: stat it directly. The Hub lookup below returns None for
|
||||
# a local path, which would drop every candidate from _recommend (file_size_mib None).
|
||||
try:
|
||||
local = Path(repo).expanduser()
|
||||
if local.is_dir():
|
||||
f = local / filename
|
||||
if f.is_file():
|
||||
return int(f.stat().st_size // (1024 * 1024))
|
||||
elif local.is_file():
|
||||
return int(local.stat().st_size // (1024 * 1024))
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
from huggingface_hub import HfApi
|
||||
info = HfApi().model_info(repo, files_metadata = True, token = os.environ.get("HF_TOKEN"))
|
||||
|
|
@ -268,6 +280,11 @@ def _compare(
|
|||
clip_sim.append(clip.image_similarity(img, ref))
|
||||
|
||||
def _mean(xs: list[float]) -> Optional[float]:
|
||||
# Preserve +inf: an identical render (reference vs itself, or a lossless
|
||||
# quant/offload) scores PSNR=inf, which is exactly the case this harness
|
||||
# verifies; dropping it as non-finite would print "-" instead of "inf".
|
||||
if xs and any(x == math.inf for x in xs):
|
||||
return math.inf
|
||||
finite = [x for x in xs if math.isfinite(x)]
|
||||
return round(sum(finite) / len(finite), 4) if finite else None
|
||||
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ import numpy as np
|
|||
|
||||
BASE = "Tongyi-MAI/Z-Image-Turbo"
|
||||
PROMPT = "A cinematic photograph of a red fox in a snowy forest at dawn, highly detailed"
|
||||
OUT = Path("/mnt/disks/unslothai/ubuntu/workspace_81/outputs/quant_research/nvfp4_images")
|
||||
OUT = Path(__file__).resolve().parent.parent / "outputs" / "quant_research" / "nvfp4_images"
|
||||
|
||||
|
||||
def _psnr(a, b):
|
||||
|
|
@ -84,11 +84,15 @@ def main(argv = None) -> int:
|
|||
p.add_argument("--seed", type = int, default = 42)
|
||||
p.add_argument("--iters", type = int, default = 3)
|
||||
p.add_argument("--min-feat", type = int, default = 512)
|
||||
p.add_argument("--out-dir", default = None, help = "image output dir (default: repo outputs/)")
|
||||
args = p.parse_args(argv)
|
||||
steps, res, seed, mf = args.steps, args.res, args.seed, args.min_feat
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
|
||||
global OUT
|
||||
if args.out_dir:
|
||||
OUT = Path(args.out_dir).expanduser()
|
||||
OUT.mkdir(parents = True, exist_ok = True)
|
||||
|
||||
def filt(mod, fqn = ""):
|
||||
|
|
|
|||
|
|
@ -25,7 +25,7 @@ import numpy as np
|
|||
|
||||
BASE = "Tongyi-MAI/Z-Image-Turbo"
|
||||
PROMPT = "A cinematic photograph of a red fox in a snowy forest at dawn, highly detailed"
|
||||
OUT = Path("/mnt/disks/unslothai/ubuntu/workspace_81/outputs/quant_research/perf_levers_images")
|
||||
OUT = Path(__file__).resolve().parent.parent / "outputs" / "quant_research" / "perf_levers_images"
|
||||
|
||||
|
||||
_LP = {"fn": None}
|
||||
|
|
@ -36,11 +36,14 @@ def _lpips(ref, arr):
|
|||
import lpips
|
||||
import torch
|
||||
|
||||
# Keep the metric model on CPU: caching it on CUDA leaves it resident across
|
||||
# variants, and each run resets peak-memory stats, so its VRAM would be charged
|
||||
# to (and reduce headroom for) every later variant's measurement.
|
||||
if _LP["fn"] is None:
|
||||
_LP["fn"] = lpips.LPIPS(net = "alex", verbose = False).cuda().eval()
|
||||
_LP["fn"] = lpips.LPIPS(net = "alex", verbose = False).eval()
|
||||
|
||||
def t(x):
|
||||
return (torch.from_numpy(x).float().permute(2, 0, 1).unsqueeze(0) / 127.5 - 1.0).cuda()
|
||||
return torch.from_numpy(x).float().permute(2, 0, 1).unsqueeze(0) / 127.5 - 1.0
|
||||
|
||||
with torch.no_grad():
|
||||
return float(_LP["fn"](t(ref), t(arr)).item())
|
||||
|
|
@ -69,6 +72,12 @@ def _reset_inductor_flags():
|
|||
ic.coordinate_descent_tuning = False
|
||||
ic.coordinate_descent_check_all_directions = False
|
||||
ic.epilogue_fusion = True
|
||||
# Reset the int-mm fusion flag too, or it leaks from the inductor_flags variant into
|
||||
# every later compiled row and the attention/fbcache measurements stop being isolated.
|
||||
try:
|
||||
ic.force_fuse_int_mm_with_mul = False
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
|
||||
|
||||
def _load():
|
||||
|
|
@ -136,6 +145,8 @@ def run(
|
|||
except Exception as exc: # noqa: BLE001
|
||||
note = f"attn({attn})={type(exc).__name__}:{str(exc)[:60]}"
|
||||
print(f" [{tag}] {note}", flush = True)
|
||||
del pipe # free the resident pipe so a skipped variant doesn't leak VRAM
|
||||
torch.cuda.empty_cache()
|
||||
return None
|
||||
if fbcache is not None:
|
||||
try:
|
||||
|
|
@ -143,6 +154,8 @@ def run(
|
|||
apply_first_block_cache(pipe.transformer, FirstBlockCacheConfig(threshold = fbcache))
|
||||
except Exception as exc: # noqa: BLE001
|
||||
print(f" [{tag}] fbcache={type(exc).__name__}:{str(exc)[:60]}", flush = True)
|
||||
del pipe
|
||||
torch.cuda.empty_cache()
|
||||
return None
|
||||
try:
|
||||
pipe.transformer.compile_repeated_blocks(fullgraph = True, dynamic = True)
|
||||
|
|
|
|||
|
|
@ -151,7 +151,12 @@ def main(argv = None) -> int:
|
|||
flush = True,
|
||||
)
|
||||
|
||||
ok = (leak_psnr == float("inf")) and (def_t < off_t) and (_psnr(off_img, def_img) >= 30)
|
||||
ok = (
|
||||
(leak_psnr == float("inf"))
|
||||
and (bal_psnr == float("inf")) # check 3: balanced must be bit-identical to off
|
||||
and (def_t < off_t)
|
||||
and (_psnr(off_img, def_img) >= 30)
|
||||
)
|
||||
print(f"\nPERF-VERIFY {'OK' if ok else 'CHECK'}", flush = True)
|
||||
return 0 if ok else 1
|
||||
|
||||
|
|
|
|||
|
|
@ -26,7 +26,7 @@ import numpy as np
|
|||
|
||||
BASE = "Tongyi-MAI/Z-Image-Turbo"
|
||||
PROMPT = "A cinematic photograph of a red fox in a snowy forest at dawn, highly detailed"
|
||||
ROOT = Path("/mnt/disks/unslothai/ubuntu/workspace_81/outputs/quant_research")
|
||||
ROOT = Path(__file__).resolve().parent.parent / "outputs" / "quant_research"
|
||||
CKPT = ROOT / "prequant_fp8" / "transformer_fp8_state.pt"
|
||||
OUT = ROOT / "prequant_images"
|
||||
MIN_FEAT = 512
|
||||
|
|
|
|||
|
|
@ -27,7 +27,7 @@ REPO = "unsloth/Z-Image-Turbo-GGUF"
|
|||
GGUF = "z-image-turbo-Q4_K_M.gguf"
|
||||
BASE = "Tongyi-MAI/Z-Image-Turbo"
|
||||
PROMPT = "A cinematic photograph of a red fox in a snowy forest at dawn, highly detailed"
|
||||
OUT = Path("/mnt/disks/unslothai/ubuntu/workspace_81/outputs/quant_research/probe_images")
|
||||
OUT = Path(__file__).resolve().parent.parent / "outputs" / "quant_research" / "probe_images"
|
||||
|
||||
|
||||
def _psnr(a, b):
|
||||
|
|
@ -39,17 +39,19 @@ _LPIPS = {"fn": None}
|
|||
|
||||
|
||||
def _lpips(ref_arr, arr):
|
||||
"""Perceptual LPIPS (alexnet) vs reference; lower is closer. None if unavailable."""
|
||||
"""Perceptual LPIPS (alexnet) vs reference; lower is closer. None if unavailable.
|
||||
|
||||
Runs on CPU so the scorer never holds CUDA memory: each row resets peak VRAM, so a
|
||||
resident GPU LPIPS module would inflate the reported load/gen VRAM and could even OOM."""
|
||||
try:
|
||||
import torch
|
||||
import lpips
|
||||
|
||||
if _LPIPS["fn"] is None:
|
||||
_LPIPS["fn"] = lpips.LPIPS(net = "alex", verbose = False).cuda().eval()
|
||||
_LPIPS["fn"] = lpips.LPIPS(net = "alex", verbose = False).eval()
|
||||
|
||||
def t(x):
|
||||
t = torch.from_numpy(x).float().permute(2, 0, 1).unsqueeze(0) / 127.5 - 1.0
|
||||
return t.cuda()
|
||||
return torch.from_numpy(x).float().permute(2, 0, 1).unsqueeze(0) / 127.5 - 1.0
|
||||
|
||||
with torch.no_grad():
|
||||
return float(_LPIPS["fn"](t(ref_arr), t(arr)).item())
|
||||
|
|
@ -111,7 +113,7 @@ def _quant_config(name):
|
|||
return MXDynamicActivationMXWeightConfig(
|
||||
activation_dtype = torch.float8_e4m3fn, weight_dtype = torch.float8_e4m3fn
|
||||
)
|
||||
except TypeError:
|
||||
except (TypeError, AttributeError):
|
||||
return MXDynamicActivationMXWeightConfig()
|
||||
raise ValueError(name)
|
||||
|
||||
|
|
|
|||
|
|
@ -1208,9 +1208,10 @@ def check_js_file(content: str, filename: str, package: str) -> list[Finding]:
|
|||
HIGH,
|
||||
package,
|
||||
filename,
|
||||
f"Python wheel ships large ({len(content) // 1024} KB) JS bundle "
|
||||
"(uncommon; manually review)",
|
||||
"",
|
||||
# Size stays in evidence, not the check label, so the baseline key
|
||||
# does not drift when a wheel's bundle grows by a few KB.
|
||||
"Python wheel ships large JS bundle (uncommon; manually review)",
|
||||
f"{len(content) // 1024} KB JS bundle",
|
||||
)
|
||||
)
|
||||
return findings
|
||||
|
|
|
|||
|
|
@ -1181,7 +1181,7 @@
|
|||
{
|
||||
"package": "tensorboard",
|
||||
"file": "tensorboard/plugins/projector/tf_projector_plugin/projector_binary.js",
|
||||
"check": "Python wheel ships large (1918 KB) JS bundle (uncommon; manually review)",
|
||||
"check": "Python wheel ships large JS bundle (uncommon; manually review)",
|
||||
"severity": "HIGH",
|
||||
"evidence": ""
|
||||
},
|
||||
|
|
|
|||
|
|
@ -19,17 +19,20 @@ from __future__ import annotations
|
|||
|
||||
import argparse
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
BACKEND = Path(__file__).resolve().parent.parent / "studio" / "backend"
|
||||
_REPO = Path(__file__).resolve().parent.parent
|
||||
_RESEARCH = _REPO / "outputs" / "quant_research"
|
||||
BACKEND = _REPO / "studio" / "backend"
|
||||
BASE = "Tongyi-MAI/Z-Image-Turbo"
|
||||
CKPT = "/mnt/disks/unslothai/ubuntu/workspace_81/outputs/quant_research/prequant_fp8/transformer_fp8.pt"
|
||||
CKPT = os.environ.get("PREQUANT_CKPT", str(_RESEARCH / "prequant_fp8" / "transformer_fp8.pt"))
|
||||
PROMPT = "A cinematic photograph of a red fox in a snowy forest at dawn, highly detailed"
|
||||
OUT = Path("/mnt/disks/unslothai/ubuntu/workspace_81/outputs/quant_research/prequant_verify_images")
|
||||
OUT = Path(os.environ.get("PREQUANT_OUT_DIR", str(_RESEARCH / "prequant_verify_images")))
|
||||
|
||||
logging.basicConfig(level = logging.INFO, format = "%(message)s")
|
||||
LOGGER = logging.getLogger("verify_prequant")
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue