perf(video): accuracy-first round 2 for HunyuanVideo-1.5: compile parity, cache quality presets, dual-GPU CFG
Cuts the shipped default's LPIPS vs the bit-exact reference from 0.224 to 0.139 while going faster (24.9 s to 21.2 s at 720p/33f/30 steps, 22.7x vs reference), and makes the remaining speed/accuracy trade a user knob. - inductor precision parity: set emulate_precision_casts=True for the regional compile (fused pointwise kernels kept fp32 intermediates where eager rounds to bf16 between ops); full-clip LPIPS vs bit-exact 0.221 to 0.052 at zero speed cost. Snapshot/restored with the other process-wide backend flags. - cache x compile composition fix: diffusers cache hooks are torch.compiler.disable'd, so every COMPUTED step ran eager (1.69 vs 1.09 s/step) under MagCache/FBCache in both enable orders. Re-point each hook's fn_ref.original_forward at a torch.compile'd wrapper of the same bound method (armed only where the speed layer compiled the block; restored before every disable_cache so the uncached path stays pristine). Balanced MagCache at 50 steps: 1.48x to 2.17x, identical skip counts, bit-identical uncached rerun after enable/disable cycles. - transformer_cache_quality knob (quality|balanced|fast; API + UI + bench) mapping to (threshold, max_skip_steps, retention_ratio). Auto resolves to the near-lossless quality preset (0.06, 2, 0.3; 1.63-1.64x at pairwise LPIPS 0.05-0.09) for the HunyuanVideo-1.5 families and to balanced (the pre-knob values, byte-identical behaviour) everywhere else. - TE auto-quant resolves dense for HunyuanVideo-1.5: TE fp8_dynamic alone moves the clip to LPIPS 0.236 vs bit-exact for zero speed win (the quantised encoder perturbs the conditioning and the trajectory amplifies it chaotically); VAE fp8 stays in auto (0.053, at the compile floor). Explicit schemes honored. - dual-GPU CFG branch parallelism (new diffusion_cfg_parallel.py): transformer proxy + DiT replica on the most-free second CUDA device + worker thread, branch-routed off the pipeline's own cache_context names. Auto engages only where measured bit-identical (eager tier: max abs diff 0.0, 1.66x); the compiled stack is explicit cfg_parallel=on (1.52x over the sequential default; per-device compiled artifacts differ by 1 bf16 ulp/step, documented in the resolved record). Fail-soft gates: family allowlist, guider CFG, pipeline kind, dense DiT, no offload, free-VRAM check; single-GPU loads are untouched and the memory plan stays single-device. - video API: the transformer_cache literal now accepts auto/magcache (an explicit magcache request was rejected at the pydantic layer); the mxfp8 family deny records the round-2 measurement (block-32 MX scaling fixes the zero-row collapse, no black frames, but is latency-neutral at LPIPS 0.37: fails both ship bars). Measured on B200 via the production lever path (video_speedmem_bench.py, which gained a --cache-quality lever and companion-quant isolation configs). Tests: 441 passing across the video inference suite (32 new for cfg-parallel, 20 for presets/arming, 3 for the inductor flag, 2 for TE auto-dense); ruff clean.
This commit is contained in:
parent
f7824b9db1
commit
7dbdd28161
15 changed files with 1929 additions and 22 deletions
|
|
@ -102,8 +102,13 @@ export interface VideoLoadRequest {
|
|||
| "sage"
|
||||
| "xformers"
|
||||
| "aiter";
|
||||
transformer_cache?: "off" | "fbcache";
|
||||
transformer_cache?: "off" | "fbcache" | "magcache";
|
||||
transformer_cache_threshold?: number;
|
||||
// Step-cache speed/accuracy preset (omit for the backend default, "balanced").
|
||||
transformer_cache_quality?: "quality" | "balanced" | "fast";
|
||||
// Dual-GPU CFG branch parallelism (omit for auto: engages on measured families when a
|
||||
// second GPU with enough free VRAM is available; bit-identical with the step cache on).
|
||||
cfg_parallel?: "off" | "auto" | "on";
|
||||
// Dense DiT precision on full-pipeline loads (omit for the hardware-ladder auto;
|
||||
// "none" pins plain bf16). GGUF / single-file checkpoints carry their own precision.
|
||||
transformer_quant?: "none" | "fp8" | "int8" | "nvfp4" | "mxfp8";
|
||||
|
|
|
|||
|
|
@ -517,7 +517,13 @@ export function VideoPage({ active = true }: { active?: boolean }) {
|
|||
const [attentionBackend, setAttentionBackend] = useState<
|
||||
"auto" | "native" | "cudnn" | "flash3" | "sage"
|
||||
>("auto");
|
||||
const [transformerCache, setTransformerCache] = useState<"auto" | "off" | "fbcache">("auto");
|
||||
const [transformerCache, setTransformerCache] = useState<"auto" | "off" | "fbcache" | "magcache">(
|
||||
"auto",
|
||||
);
|
||||
const [cacheQuality, setCacheQuality] = useState<"auto" | "quality" | "balanced" | "fast">(
|
||||
"auto",
|
||||
);
|
||||
const [cfgParallel, setCfgParallel] = useState<"auto" | "off" | "on">("auto");
|
||||
const [transformerQuant, setTransformerQuant] = useState<
|
||||
"auto" | "none" | "fp8" | "int8" | "nvfp4" | "mxfp8"
|
||||
>("auto");
|
||||
|
|
@ -939,7 +945,9 @@ export function VideoPage({ active = true }: { active?: boolean }) {
|
|||
speed_mode: speedMode === "auto" ? undefined : speedMode,
|
||||
attention_backend: attentionBackend === "auto" ? undefined : attentionBackend,
|
||||
transformer_cache: transformerCache === "auto" ? undefined : transformerCache,
|
||||
transformer_cache_quality: cacheQuality === "auto" ? undefined : cacheQuality,
|
||||
transformer_quant: transformerQuant === "auto" ? undefined : transformerQuant,
|
||||
cfg_parallel: cfgParallel === "auto" ? undefined : cfgParallel,
|
||||
});
|
||||
} catch (err) {
|
||||
dismissLoadToast();
|
||||
|
|
@ -959,7 +967,9 @@ export function VideoPage({ active = true }: { active?: boolean }) {
|
|||
speedMode,
|
||||
attentionBackend,
|
||||
transformerCache,
|
||||
cacheQuality,
|
||||
transformerQuant,
|
||||
cfgParallel,
|
||||
],
|
||||
);
|
||||
|
||||
|
|
@ -1233,7 +1243,7 @@ export function VideoPage({ active = true }: { active?: boolean }) {
|
|||
/>
|
||||
<AdvancedSelect
|
||||
label="Step cache"
|
||||
hint="First-Block-Cache reuses the transformer tail across steps for many-step models. Auto turns it on at 20+ steps and off for few-step distilled models, re-checked per clip."
|
||||
hint="Reuses transformer work across denoise steps for many-step models. Auto picks the family's measured mode (MagCache on HunyuanVideo-1.5, First-Block-Cache elsewhere) at 20+ steps and turns it off for few-step distilled models, re-checked per clip."
|
||||
badge={<ResolvedBadge status={status} controlKey="transformer_cache" />}
|
||||
value={transformerCache}
|
||||
onValueChange={(v) => setTransformerCache(v as typeof transformerCache)}
|
||||
|
|
@ -1241,6 +1251,32 @@ export function VideoPage({ active = true }: { active?: boolean }) {
|
|||
["auto", "Auto"],
|
||||
["off", "Off"],
|
||||
["fbcache", "First-Block-Cache"],
|
||||
["magcache", "MagCache"],
|
||||
]}
|
||||
/>
|
||||
<AdvancedSelect
|
||||
label="Cache quality"
|
||||
hint="Speed/accuracy preset for the step cache. Quality skips conservatively (near-lossless, smaller speedup); Balanced skips more for more speed; Fast is the most aggressive. Auto picks the family's measured default (Quality on HunyuanVideo-1.5, Balanced elsewhere)."
|
||||
badge={<ResolvedBadge status={status} controlKey="transformer_cache_quality" />}
|
||||
value={cacheQuality}
|
||||
onValueChange={(v) => setCacheQuality(v as typeof cacheQuality)}
|
||||
options={[
|
||||
["auto", "Auto"],
|
||||
["quality", "Quality"],
|
||||
["balanced", "Balanced"],
|
||||
["fast", "Fast"],
|
||||
]}
|
||||
/>
|
||||
<AdvancedSelect
|
||||
label="Dual-GPU CFG"
|
||||
hint="Runs the two guidance branches concurrently, one on a DiT replica on a second GPU (~1.7x on HunyuanVideo-1.5; the replica takes ~20 GB VRAM there). Auto engages only where the output is bit-identical to single-GPU (the Eager speed tier); On also parallelizes the compiled stack, accepting fp-noise-level divergence."
|
||||
badge={<ResolvedBadge status={status} controlKey="cfg_parallel" />}
|
||||
value={cfgParallel}
|
||||
onValueChange={(v) => setCfgParallel(v as typeof cfgParallel)}
|
||||
options={[
|
||||
["auto", "Auto"],
|
||||
["off", "Off"],
|
||||
["on", "On"],
|
||||
]}
|
||||
/>
|
||||
{status?.loaded && canReapply && (
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue