Extend fp8 black-frame deny to HunyuanVideo-1.5; keep LTX-2 on fp8

Measured the fp8 DiT auto-quant path across the remaining dense-pipeline video families on
B200 (production torch._scaled_mm per-row fp8, no MSLK):
  - HunyuanVideo-1.5 (480p + 720p repacks): every frame black (mean luma 0.0, LPIPS 0.82);
    int8 is clean (mean 102.7 vs dense 99.9). Same failure as Wan / qwen-image.
  - LTX-2: fp8 renders clean (mean 153.7, matches int8's 157.7) -- NOT a black-frame family.

So deny fp8/mxfp8/nvfp4 for hunyuanvideo-1.5 and hunyuanvideo-1.5-720p (fall to int8), and
deliberately leave LTX-2 on fp8. The deny stays measured per family, not a blanket video rule:
a blanket deny would have wrongly forced LTX-2 off fp8. Adds a Hunyuan deny test that also
asserts LTX-2 keeps fp8; 49/49 transformer-quant tests pass.

video_speedmem_bench.py gains guidance_via_guider support (HunyuanVideo-1.5 sets CFG on a
guider component and its __call__ takes no guidance_scale / callback_on_step_end), so the
harness can drive Hunyuan the same way the loader does.
This commit is contained in:
Daniel Han 2026-07-08 15:12:00 +00:00
commit c947b33ef8
3 changed files with 56 additions and 27 deletions

View file

@ -310,7 +310,7 @@ def _apply_levers(pipe, cfg: dict, *, fam_name: str, fam_obj, force_fp32_vae: bo
def _timed_video(pipe, *, steps, width, height, num_frames, guidance, seed, cache_mode,
dit_quant_active, default_steps, logger=None):
dit_quant_active, default_steps, guidance_via_guider=False, logger=None):
"""One clip generation. Re-checks FBCache per generation (maybe_toggle_step_cache) exactly
like the loader, then times total + per-step. Returns (output, total_s, [per_step_ms])."""
import torch
@ -337,18 +337,26 @@ def _timed_video(pipe, *, steps, width, height, num_frames, guidance, seed, cach
last[0] = now
return kw
kwargs = dict(
prompt=PROMPT, width=width, height=height, num_frames=num_frames,
num_inference_steps=steps, generator=g,
)
if guidance_via_guider:
# HunyuanVideo-1.5: CFG lives on a guider component and __call__ takes no
# guidance_scale / callback_on_step_end (the loader writes the scale onto pipe.guider).
guider = getattr(pipe, "guider", None)
if guider is not None and hasattr(guider, "guidance_scale"):
try:
guider.guidance_scale = guidance
except Exception:
pass
else:
kwargs["guidance_scale"] = guidance
kwargs["callback_on_step_end"] = _cb
_sync()
t0 = time.perf_counter()
out = pipe(
prompt=PROMPT,
width=width,
height=height,
num_frames=num_frames,
num_inference_steps=steps,
guidance_scale=guidance,
generator=g,
callback_on_step_end=_cb,
)
out = pipe(**kwargs)
_sync()
return out, (time.perf_counter() - t0), step_ts
@ -365,6 +373,7 @@ def _run_config(name: str, cfg: dict, *, family: str, steps: int, width: int, he
guidance = spec.get("guidance", 5.0)
fam_obj = detect_video_family(repo)
default_steps = getattr(fam_obj, "default_steps", 50)
gvg = bool(getattr(fam_obj, "guidance_via_guider", False))
_empty(); _reset_peak()
pipe = _build_pipe(repo, force_fp32)
@ -382,7 +391,7 @@ def _run_config(name: str, cfg: dict, *, family: str, steps: int, width: int, he
_timed_video(
pipe, steps=steps, width=width, height=height, num_frames=num_frames, guidance=guidance,
seed=seed, cache_mode=cache_mode, dit_quant_active=dit_active, default_steps=default_steps,
logger=logger,
guidance_via_guider=gvg, logger=logger,
)
_reset_peak()
dts, steps_ms = [], []
@ -391,7 +400,7 @@ def _run_config(name: str, cfg: dict, *, family: str, steps: int, width: int, he
last_out, dt, st = _timed_video(
pipe, steps=steps, width=width, height=height, num_frames=num_frames, guidance=guidance,
seed=seed, cache_mode=cache_mode, dit_quant_active=dit_active, default_steps=default_steps,
logger=logger,
guidance_via_guider=gvg, logger=logger,
)
dts.append(dt)
steps_ms.append(_median(st) if st else 0.0)

View file

@ -133,20 +133,23 @@ _AUTO_LADDER: tuple[tuple[tuple[int, int], tuple[str, ...]], ...] = (
# qwen-image + mxfp8 -> real semantic damage at 1024px (CLIP delta mean 0.0146, worst
# cases 0.064 / 0.102 -- 2x the per-case bound).
# qwen-image + nvfp4 -> LPIPS mean 0.51 vs bf16: unusable.
# wan2.2 + fp8 -> every frame black (mean luma 0.0000, LPIPS ~0.80 vs bf16),
# reproduced on B200 at 512x320 and 704x480 with the production
# torch._scaled_mm per-row fp8 path (no MSLK): the Wan DiT's
# activation outliers exceed per-row fp8's range, the same failure
# mode as qwen-image. int8 dynamic (per-token) is clean on Wan
# (non-black, correct contrast; First-Block-Cache engages normally
# instead of over-caching the degenerate black activations).
# int8 dynamic (per-token) is excellent on Qwen (LPIPS mean 0.069 / SSIM 0.958) and clean on
# Wan, so the auto ladder falls through to it. mxfp8 / nvfp4 are denied alongside fp8 on the
# Wan families conservatively (the same per-block scaled_mm family as the confirmed-black fp8,
# and mxfp8 is a prototype) so auto lands on the battle-tested int8; they can be re-enabled per
# family once separately validated in-bar, like the nvfp4 auto-ladder TODO. The deny also
# applies to an EXPLICIT request: a scheme that renders black frames has no legitimate use, and
# returning None gives the caller the same fallback contract as an unsupported scheme (GGUF).
# wan2.2 / hunyuanvideo-1.5 + fp8 -> every frame black (mean luma 0.0000, LPIPS ~0.80-0.82
# vs bf16), reproduced on B200 with the production torch._scaled_mm
# per-row fp8 path (no MSLK): the DiT's activation outliers exceed
# per-row fp8's range, the same failure mode as qwen-image. int8
# dynamic (per-token) is clean on both (non-black, correct contrast;
# First-Block-Cache engages normally instead of over-caching the
# degenerate black activations).
# NOTE this is NOT universal across video: ltx-2 + fp8 renders clean (mean 153.7, non-black) on
# the same stack, so it is deliberately NOT denied -- the deny is measured per family, not a
# blanket video rule. int8 dynamic (per-token) is excellent on Qwen (LPIPS mean 0.069 / SSIM
# 0.958) and clean on Wan / Hunyuan, so the auto ladder falls through to it there. mxfp8 / nvfp4
# are denied alongside fp8 on the black-frame families conservatively (the same per-block
# scaled_mm family as the confirmed-black fp8, and mxfp8 is a prototype) so auto lands on the
# battle-tested int8; they can be re-enabled per family once separately validated in-bar, like
# the nvfp4 auto-ladder TODO. The deny also applies to an EXPLICIT request: a scheme that renders
# black frames has no legitimate use, and returning None gives the caller the same fallback
# contract as an unsupported scheme (GGUF build).
_FAMILY_SCHEME_DENY: dict[str, frozenset[str]] = {
"qwen-image": frozenset({TQ_FP8, TQ_MXFP8, TQ_NVFP4}),
"qwen-image-edit": frozenset({TQ_FP8, TQ_MXFP8, TQ_NVFP4}), # same DiT + activations
@ -154,6 +157,11 @@ _FAMILY_SCHEME_DENY: dict[str, frozenset[str]] = {
# 5B TI2V and the A14B MoE share the DiT class + activation profile, so both deny -> int8.
"wan2.2-ti2v-5b": frozenset({TQ_FP8, TQ_MXFP8, TQ_NVFP4}),
"wan2.2-t2v-a14b": frozenset({TQ_FP8, TQ_MXFP8, TQ_NVFP4}),
# HunyuanVideo-1.5 DiT (HunyuanVideo15Transformer3DModel): fp8 renders black frames (measured,
# LPIPS 0.82); int8 is clean. The 480p and 720p repacks share the DiT + activations, so both
# deny -> int8. (ltx-2 fp8 measures clean on the same stack, so it is intentionally absent.)
"hunyuanvideo-1.5": frozenset({TQ_FP8, TQ_MXFP8, TQ_NVFP4}),
"hunyuanvideo-1.5-720p": frozenset({TQ_FP8, TQ_MXFP8, TQ_NVFP4}),
}

View file

@ -575,6 +575,18 @@ def test_family_deny_refuses_explicit_fp8_for_wan(monkeypatch):
assert select_transformer_quant_scheme(_target(), "int8", family = "wan2.2-ti2v-5b") == TQ_INT8
def test_family_deny_auto_skips_fp8_for_hunyuan(monkeypatch):
# HunyuanVideo-1.5 DiT also renders black frames on fp8 (measured, LPIPS 0.82); both the
# 480p and 720p repacks deny fp8/mxfp8/nvfp4 and fall to int8. ltx-2 is NOT denied (its
# fp8 measures clean), so the deny is per family, not a blanket video rule.
_stub_torch(monkeypatch, cc = (10, 0))
_allow(monkeypatch, {TQ_FP8, TQ_NVFP4, TQ_MXFP8, TQ_INT8})
assert select_transformer_quant_scheme(_target(), "auto", family = "hunyuanvideo-1.5") == TQ_INT8
assert select_transformer_quant_scheme(_target(), "auto", family = "hunyuanvideo-1.5-720p") == TQ_INT8
# ltx-2 keeps the ladder head (fp8) -- it is not a black-frame family.
assert select_transformer_quant_scheme(_target(), "auto", family = "ltx-2") == TQ_FP8
def test_family_deny_no_family_keeps_ladder(monkeypatch):
# Without a family (or an unknown one) the ladder is unchanged: fp8 first on B200.
_stub_torch(monkeypatch, cc = (10, 0))