The Wan fp8 black frame was root-caused (scripts/fp8_layer_ablation.py, measured on B200 with the production torch._scaled_mm path): per-row fp8 scales each activation row by row_amax/448, and the text prompt is padded to 512 tokens (~all padding for a short prompt), so condition_embedder's text embedder divides a zero padding row by a zero scale, which infs and renders every frame black. That embedder's bias makes every downstream row non-zero, so the whole 30-block attn1/attn2/ffn stack is fp8-clean (fp8-except- condition_embedder measured cosine 0.9998 vs bf16, 0 non-finite; fp8- everywhere is 100% non-finite). So the blanket fp8 deny was heavier than needed for Wan. Remove fp8 from the Wan deny and keep only condition_embedder in bf16 via a new _FP8_FAMILY_EXCLUDE_NAME_TOKENS; auto now restores fp8 (the Blackwell ladder head) for Wan2.2-TI2V-5B and -T2V-A14B (shared DiT class and padded-text conditioning). Full-generation check (512x320, 25 frames, 30 steps, cache on and off): mixed-fp8 is non-black (mean luma 182.6 vs dense 181.2), more accurate than int8 (LPIPS 0.129 vs 0.180 no-cache, 0.224 vs 0.251 with FBCache), faster (49.9 vs 64.6 ms/step; int8 was a per-step regression vs the 59.8 ms/step dense), at the same memory (19.34 GB, both -20% vs dense). HunyuanVideo-1.5 keeps the fp8 deny: its MMDiT masks the padding text tokens to zero inside every block, so the per-block context stream (add_*_proj / to_add_out / ff_context) regenerates zero rows layer after layer (fp8 on only the main blocks is 100% non-finite) so no small exclude set exists and int8 stays. mxfp8 / nvfp4 remain denied for Wan (same per-row scaled_mm family, not separately validated). exclude_tokens_for_scheme now takes an optional family, threaded through the runtime quantiser and the offline prequant builder + validator so offline == runtime (a stale Wan fp8 checkpoint baked without the exclude is rejected and re-quantised rather than loaded). Adds scripts/fp8_layer_ablation.py (the per-layer ablation probe) and a mean-luma black-frame metric plus mixed-fp8 vs int8 configs to the video bench.
174 lines
7.7 KiB
Python
174 lines
7.7 KiB
Python
# SPDX-License-Identifier: AGPL-3.0-only
|
|
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
|
|
|
|
"""Build a pre-quantized transformer checkpoint for the Studio diffusion fast path.
|
|
|
|
Quantise a model's dense bf16 DiT transformer ONCE and save the quantized state dict, so
|
|
the backend can load the already-quantized weights at runtime (meta-init +
|
|
load_state_dict(assign=True)) instead of materialising the dense bf16 on the GPU. That
|
|
drops the transformer GPU load peak ~2x and the download ~2x for fp8 (measured on Z-Image:
|
|
12.9 -> 6.3 GB peak, 12 -> 6.28 GB on disk), with bit-identical output -- it is the exact
|
|
same torchao config + min_features filter the runtime path uses, applied ahead of time.
|
|
|
|
Run on one CUDA (Blackwell / Ada / Hopper) GPU. fp8 works on torch 2.9+; the FP4/MX schemes
|
|
need the newer kernels (see scripts/nvfp4_t211_probe.py).
|
|
|
|
python scripts/build_prequant_checkpoint.py \
|
|
--base Tongyi-MAI/Z-Image-Turbo --family z-image --scheme fp8 \
|
|
--out outputs/quant_research/prequant_fp8/transformer_fp8.pt [--upload-repo ORG/REPO]
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import sys
|
|
import time
|
|
from pathlib import Path
|
|
|
|
BACKEND = Path(__file__).resolve().parent.parent / "studio" / "backend"
|
|
|
|
|
|
def main(argv = None) -> int:
|
|
p = argparse.ArgumentParser()
|
|
p.add_argument(
|
|
"--base", required = True, help = "diffusers base repo (carries the transformer subfolder)"
|
|
)
|
|
p.add_argument("--family", required = True, help = "diffusion family name/alias (e.g. z-image)")
|
|
p.add_argument("--scheme", required = True, help = "quant scheme: int8 | fp8 | nvfp4 | mxfp8")
|
|
p.add_argument("--out", required = True, help = "output .pt path for the checkpoint")
|
|
p.add_argument("--min-features", type = int, default = 512)
|
|
p.add_argument("--dtype", default = "bfloat16", choices = ["bfloat16"])
|
|
p.add_argument("--hf-token", default = None)
|
|
p.add_argument(
|
|
"--upload-repo", default = None, help = "optional HF repo id to upload the checkpoint to"
|
|
)
|
|
p.add_argument("--upload-revision", default = None)
|
|
args = p.parse_args(argv)
|
|
|
|
sys.path.insert(0, str(BACKEND))
|
|
import torch
|
|
import torchao
|
|
import diffusers
|
|
|
|
from core.inference.diffusion_families import detect_family
|
|
from core.inference.diffusion_prequant import PREQUANT_FORMAT, prequant_filename
|
|
|
|
# Reuse the runtime quant factory + filter so offline == runtime (the LPIPS-0 invariant).
|
|
from core.inference.diffusion_transformer_quant import (
|
|
FP8_GRANULARITY,
|
|
TQ_FP8,
|
|
TQ_SCHEMES,
|
|
_REQUIRE_BF16_SCHEMES,
|
|
_make_quant_config,
|
|
_resolve_fast_accum,
|
|
exclude_tokens_for_scheme,
|
|
make_filter_fn,
|
|
)
|
|
from torchao.quantization import quantize_
|
|
|
|
scheme = args.scheme.strip().lower()
|
|
if scheme not in TQ_SCHEMES:
|
|
print(f"error: --scheme must be one of {TQ_SCHEMES} (not 'auto')", flush = True)
|
|
return 2
|
|
fam = detect_family(args.base, override = args.family)
|
|
if fam is None:
|
|
print(f"error: unknown family '{args.family}'", flush = True)
|
|
return 2
|
|
transformer_cls = getattr(diffusers, fam.transformer_class)
|
|
|
|
print(f"== build prequant ({fam.name}/{scheme}, min_feat={args.min_features}) ==", flush = True)
|
|
print(f" loading dense transformer from {args.base} (subfolder=transformer) ...", flush = True)
|
|
t0 = time.time()
|
|
transformer = transformer_cls.from_pretrained(
|
|
args.base, subfolder = "transformer", torch_dtype = torch.bfloat16, token = args.hf_token
|
|
).to("cuda")
|
|
print(f" quantising in place ({scheme}) ...", flush = True)
|
|
# Mirror the runtime path EXACTLY (the offline == runtime, LPIPS-0 invariant): for int8 also
|
|
# skip the M=1 AdaLN-modulation / conditioning-embedder projections, else the saved checkpoint
|
|
# bakes them as int8 and crashes (torch._int_mm needs M>16) at the first denoise step on
|
|
# Flux / Qwen. fp8 / fp4 / mx use scaled_mm (no M limit) and exclude nothing EXCEPT on families
|
|
# whose padded conditioning would divide by a zero row scale (e.g. Wan's condition_embedder);
|
|
# pass the family so the offline set matches the runtime one exactly.
|
|
exclude_name_tokens = exclude_tokens_for_scheme(scheme, fam.name)
|
|
# fp8 and mxfp8 assert a bf16 weight, so their filter must skip any non-bf16 Linear the
|
|
# transformer keeps: a mixed-precision DiT (Wan / Hunyuan) retains its _keep_in_fp32_modules in
|
|
# fp32 even under torch_dtype=bf16, so quantising one would raise inside quantize_ and abort the
|
|
# whole pass. nvfp4 quantises fp32 fine, so it is not gated. Runtime quantize_transformer gates
|
|
# this on scheme membership; mirror it here so the offline checkpoint quantises the exact same
|
|
# layer set (offline == runtime).
|
|
require_bf16 = scheme in _REQUIRE_BF16_SCHEMES
|
|
# fp8 bakes the accumulate mode into the saved kernels; record the resolved choice so the
|
|
# loader can refuse a checkpoint whose baked value contradicts an explicit runtime request.
|
|
fast_accum = _resolve_fast_accum(None) if scheme == TQ_FP8 else None
|
|
quantize_(
|
|
transformer,
|
|
_make_quant_config(scheme),
|
|
filter_fn = make_filter_fn(
|
|
args.min_features,
|
|
exclude_name_tokens = exclude_name_tokens,
|
|
require_bf16 = require_bf16,
|
|
),
|
|
)
|
|
|
|
# Move the state dict to CPU for a portable, GPU-free artifact.
|
|
state_dict = {
|
|
k: (v.detach().to("cpu") if hasattr(v, "detach") else v)
|
|
for k, v in transformer.state_dict().items()
|
|
}
|
|
metadata = {
|
|
"base_model_id": args.base,
|
|
"family": fam.name,
|
|
"scheme": scheme,
|
|
"min_features": args.min_features,
|
|
# The layers skipped for this scheme (int8's M=1 modulation projections; () for
|
|
# the scaled_mm schemes), whether non-bf16 Linears were skipped (the scaled_mm
|
|
# bf16 gate), and, for fp8, the baked accumulate mode. All let the loader reject a
|
|
# checkpoint that would not match the runtime path.
|
|
"exclude_name_tokens": list(exclude_name_tokens),
|
|
"require_bf16": require_bf16,
|
|
"fast_accum": fast_accum,
|
|
"torch_dtype": args.dtype,
|
|
"quant_backend": "torchao",
|
|
"transformer_class": fam.transformer_class,
|
|
"torch_version": torch.__version__,
|
|
"torchao_version": getattr(torchao, "__version__", "?"),
|
|
"diffusers_version": diffusers.__version__,
|
|
}
|
|
# Record the fp8 granularity so the loader can reject a stale per-tensor checkpoint
|
|
# (the runtime now requires per-row; see FP8_GRANULARITY).
|
|
if scheme == TQ_FP8:
|
|
metadata["fp8_granularity"] = FP8_GRANULARITY
|
|
ckpt = {
|
|
"format": PREQUANT_FORMAT,
|
|
"metadata": metadata,
|
|
"state_dict": state_dict,
|
|
}
|
|
|
|
out = Path(args.out)
|
|
out.parent.mkdir(parents = True, exist_ok = True)
|
|
torch.save(ckpt, out)
|
|
size_gb = out.stat().st_size / 1e9
|
|
print(f" saved {out} ({size_gb:.2f} GB) in {time.time() - t0:.0f}s", flush = True)
|
|
print(f" metadata: {ckpt['metadata']}", flush = True)
|
|
|
|
if args.upload_repo:
|
|
from huggingface_hub import HfApi
|
|
|
|
dest = prequant_filename(scheme)
|
|
print(f" uploading -> {args.upload_repo}:{dest} ...", flush = True)
|
|
api = HfApi(token = args.hf_token)
|
|
api.create_repo(args.upload_repo, exist_ok = True)
|
|
api.upload_file(
|
|
path_or_fileobj = str(out),
|
|
path_in_repo = dest,
|
|
repo_id = args.upload_repo,
|
|
revision = args.upload_revision,
|
|
)
|
|
print(f" uploaded {dest} to {args.upload_repo}", flush = True)
|
|
|
|
print("BUILD-PREQUANT-DONE", flush = True)
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|