Studio diffusion (Phase 8) review fixes: quant compile + nvfp4 path
- diffusion: a torchao-quantized transformer is committed only compiled. A dense model resolves to speed_mode=off, which would run the quant eager (~30x slower than the GGUF it replaced), so when transformer_quant engaged and speed resolved to off, promote to default (regional compile); warn loudly if compile still does not engage. - diffusion_transformer_quant: build the nvfp4 config with use_triton_kernel=False so the CUTLASS FP4 path is used (torchao defaults to the Triton kernel, which needs MSLK); otherwise the smoke probe fails on CUTLASS-only Blackwell and silently drops to GGUF. - nvfp4_probe: repo-relative output dir + --out-dir (was an author-absolute /mnt path). - test asserts the eager-quant -> default-compile promotion.
This commit is contained in:
parent
daad898d05
commit
8bdd307889
4 changed files with 39 additions and 2 deletions
|
|
@ -16,7 +16,7 @@ import numpy as np
|
|||
|
||||
BASE = "Tongyi-MAI/Z-Image-Turbo"
|
||||
PROMPT = "A cinematic photograph of a red fox in a snowy forest at dawn, highly detailed"
|
||||
OUT = Path("/mnt/disks/unslothai/ubuntu/workspace_81/outputs/quant_research/nvfp4_images")
|
||||
OUT = Path(__file__).resolve().parent.parent / "outputs" / "quant_research" / "nvfp4_images"
|
||||
|
||||
|
||||
def _psnr(a, b):
|
||||
|
|
@ -84,11 +84,15 @@ def main(argv = None) -> int:
|
|||
p.add_argument("--seed", type = int, default = 42)
|
||||
p.add_argument("--iters", type = int, default = 3)
|
||||
p.add_argument("--min-feat", type = int, default = 512)
|
||||
p.add_argument("--out-dir", default = None, help = "image output dir (default: repo outputs/)")
|
||||
args = p.parse_args(argv)
|
||||
steps, res, seed, mf = args.steps, args.res, args.seed, args.min_feat
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
|
||||
global OUT
|
||||
if args.out_dir:
|
||||
OUT = Path(args.out_dir).expanduser()
|
||||
OUT.mkdir(parents = True, exist_ok = True)
|
||||
|
||||
def filt(mod, fqn = ""):
|
||||
|
|
|
|||
|
|
@ -45,6 +45,7 @@ from .diffusion_memory import (
|
|||
snapshot_device_memory,
|
||||
)
|
||||
from .diffusion_speed import (
|
||||
SPEED_DEFAULT,
|
||||
SPEED_OFF,
|
||||
apply_speed_optims,
|
||||
resolve_speed_mode,
|
||||
|
|
@ -534,6 +535,18 @@ class DiffusionBackend:
|
|||
# the quant noise floor), dense models stay bit-identical `off`. An
|
||||
# explicit speed_mode (incl. "off") is honored verbatim.
|
||||
effective_speed = resolve_speed_mode(speed_mode, is_gguf = bool(gguf_filename))
|
||||
# A torchao-quantized dense transformer runs its matmuls through the
|
||||
# regional torch.compile; UNcompiled (eager) it is ~30x slower and would
|
||||
# lose to the GGUF fallback. A dense model otherwise resolves to `off`, so
|
||||
# force at least `default` (regional compile) whenever the quant engaged,
|
||||
# or the opt-in "fast" path silently commits an eager, pathologically slow
|
||||
# pipeline.
|
||||
if transformer_quant_engaged is not None and effective_speed == SPEED_OFF:
|
||||
logger.info(
|
||||
"diffusion.transformer_quant: forcing speed_mode=default "
|
||||
"(quantized transformer must be compiled; eager is ~30x slower)"
|
||||
)
|
||||
effective_speed = SPEED_DEFAULT
|
||||
# Opt-in speed optims run BEFORE placement (channels_last / compile
|
||||
# must precede CPU offload). Snapshot the process-wide backend flags
|
||||
# first so unload can restore them: TF32 / cudnn.benchmark are global,
|
||||
|
|
@ -547,6 +560,16 @@ class DiffusionBackend:
|
|||
speed_mode = effective_speed,
|
||||
logger = logger,
|
||||
)
|
||||
if transformer_quant_engaged is not None and not speed_applied.get("compiled"):
|
||||
# Promotion above could not engage compile (e.g. the family is not
|
||||
# compile-friendly, or compile_repeated_blocks failed): the quantized
|
||||
# transformer is now running eager, which is far slower than the GGUF
|
||||
# path it replaced. Surface it loudly rather than hiding the regression.
|
||||
logger.warning(
|
||||
"diffusion.transformer_quant: %s engaged but the transformer is NOT "
|
||||
"compiled; eager torchao quant is ~30x slower than GGUF here",
|
||||
transformer_quant_engaged,
|
||||
)
|
||||
# Quantise the dense companion text encoder(s) (opt-in fp8 / nvfp4),
|
||||
# also before placement so the offload hooks move the smaller weights.
|
||||
te_quant = quantize_text_encoders(
|
||||
|
|
|
|||
|
|
@ -257,7 +257,14 @@ def _make_quant_config(scheme: str, fast_accum: Optional[bool] = None) -> Any:
|
|||
return Float8DynamicActivationFloat8WeightConfig()
|
||||
if scheme == TQ_NVFP4:
|
||||
from torchao.prototype.mx_formats import NVFP4DynamicActivationNVFP4WeightConfig
|
||||
return NVFP4DynamicActivationNVFP4WeightConfig()
|
||||
# Select the CUTLASS FP4 path, not the default Triton kernel: torchao defaults
|
||||
# use_triton_kernel=True, which needs MSLK installed. On a Blackwell box with the
|
||||
# CUTLASS FP4 extension but no MSLK, the default would make the smoke probe fail
|
||||
# and silently fall back to GGUF instead of using the FP4 tensor cores.
|
||||
try:
|
||||
return NVFP4DynamicActivationNVFP4WeightConfig(use_triton_kernel = False)
|
||||
except TypeError: # older torchao without the knob
|
||||
return NVFP4DynamicActivationNVFP4WeightConfig()
|
||||
if scheme == TQ_MXFP8:
|
||||
import torch
|
||||
from torchao.prototype.mx_formats import MXDynamicActivationMXWeightConfig
|
||||
|
|
|
|||
|
|
@ -947,6 +947,9 @@ def test_transformer_quant_dense_path_engaged(fake_runtime, tmp_path, monkeypatc
|
|||
transformer_quant = "fp8",
|
||||
)
|
||||
assert status["transformer_quant"] == "fp8"
|
||||
# No speed_mode was given, but a quantized transformer is ~30x slower eager, so the
|
||||
# backend promotes it to `default` (regional compile) instead of the dense `off`.
|
||||
assert status["speed_mode"] == "default"
|
||||
assert calls["from_pretrained"] == 1 and calls["quantize"] == 1
|
||||
assert calls["quant_mode"] == "fp8"
|
||||
assert calls["fp_kwargs"]["subfolder"] == "transformer" # dense transformer subfolder
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue