Add the HunyuanImage 2.1 family to the image backend

The hunyuanvideo-community diffusers mirror carries the full stack in
standard layout: a 17B dual-stream DiT (32.5 GB bf16), a Qwen2.5-VL text
encoder, a ByT5 glyph encoder, the 32x HunyuanImage VAE, and
guider/ocr_guider components (AdaptiveProjectedMixGuidance) that diffusers
0.39 loads natively, so the generic from_pretrained pipeline path covers
everything with no per-component assembly.

Family notes:
- The call's guidance knob is distilled_guidance_scale (there is no
  guidance_scale kwarg), so cfg_kwarg routes the UI value there; real CFG
  runs inside the repo's guider at its baked scale. Defaults follow the
  card recipe: 50 steps, 3.25.
- 2K-native: verified live at both 1024 and 2048.
- Coexists with the HunyuanImage-3.0 structured exclusion (3.0 has no
  diffusers pipeline and stays excluded with its stated reason).
- int8/fp8 dense quantization verified live (LPIPS 0.186 both vs same-seed
  bf16); a short prompt does not trip the int8 torch._int_mm minimum on
  this arch, so no family exclude entry is needed.
- bf16 component table for the memory planner: (32.5, 16.3, 0.8) GB.
This commit is contained in:
Daniel Han 2026-07-17 12:52:01 +00:00
commit a3992cd0b6
4 changed files with 84 additions and 0 deletions

View file

@ -300,6 +300,9 @@ _TRUSTED_NON_GGUF_REPOS = frozenset(
# Lumina Image 2.0: standard diffusers layout (Gemma2-2B encoder), safetensors-only,
# loads through the generic from_pretrained pipeline path.
"alpha-vllm/lumina-image-2.0",
# HunyuanImage 2.1: the community diffusers mirror (open, tencent-hunyuan-community
# license), safetensors-only, including the diffusers-native guider components.
"hunyuanvideo-community/hunyuanimage-2.1-diffusers",
# Ideogram 4: no bf16 ships. -fp8 stores the two DiTs as raw float8 (the family base);
# the two nf4 repos are identical bnb-4bit exports (both listed so either id loads).
"ideogram-ai/ideogram-4-fp8",

View file

@ -52,6 +52,8 @@ _FAMILY_BF16_GB: dict[str, tuple[float, float, float]] = {
"krea-2": (26.3, 8.9, 0.5),
# Ships fp32 (10.4 + 10.5 + 0.3 GB of shards); bf16-resident is half.
"lumina-2": (5.2, 5.2, 0.2),
# 17B dual-stream DiT (32.5 GB bf16 on disk) + Qwen2.5-VL 15.5 GB + ByT5 0.8 GB.
"hunyuanimage-2.1": (32.5, 16.3, 0.8),
# Two ~9.3B DiTs (conditional + unconditional_transformer for Ideogram's dual-branch CFG),
# both resident, plus a Qwen3-VL encoder. The vendor stores them as raw float8; these are the
# bf16-resident sizes after the dtype cast, so each doubles (37.2 = 2 x 18.6, encoder 16.3).

View file

@ -329,6 +329,24 @@ _FAMILIES: tuple[DiffusionFamily, ...] = (
# Published and validated bf16-only upstream; keep the fp16 fallback off like z-image.
fp16_incompatible = True,
),
# HunyuanImage 2.1 (diffusers >= 0.39): a 17B dual-stream DiT with a Qwen2.5-VL text
# encoder, a ByT5 glyph encoder, and the 32x-compression HunyuanImage VAE. The community
# mirror also ships guider/ocr_guider components (AdaptiveProjectedMixGuidance), which
# 0.39 loads natively, so the generic from_pretrained pipeline path covers the whole
# stack. 2K-native (the card recipe renders 2048x2048); classifier-free guidance runs
# inside the repo's guider at its baked scale, and the call's own guidance knob is
# distilled_guidance_scale (there is no guidance_scale kwarg). Distinct from
# HunyuanImage-3.0, which stays excluded above: 2.1 has a real diffusers pipeline.
DiffusionFamily(
name = "hunyuanimage-2.1",
pipeline_class = "HunyuanImagePipeline",
transformer_class = "HunyuanImageTransformer2DModel",
base_repo = "hunyuanvideo-community/HunyuanImage-2.1-Diffusers",
cfg_kwarg = "distilled_guidance_scale",
aliases = ("hunyuanimage-2.1-diffusers", "hunyuanimage2.1"),
# Exported bf16-only; keep the fp16 fallback off like z-image / krea-2.
fp16_incompatible = True,
),
# Ideogram 4 (diffusers >= 0.39): a 34-layer DiT PAIR (conditional + unconditional_transformer
# for dual-branch CFG, both ~9B, so memory planning counts two DiTs) with a Qwen3-VL encoder.
# No bf16 checkpoint: ideogram-4-fp8 (raw float8, upcast on load) is the highest-precision
@ -511,6 +529,9 @@ _GENERATION_DEFAULTS: tuple[tuple[str, int, float], ...] = (
# Lumina Image 2.0 model-card: 50 steps, guidance 4 (plus cfg_trunc_ratio 0.25, which the
# loader passes itself; see LUMINA2_FAMILY_NAME).
("lumina", 50, 4.0),
# HunyuanImage 2.1 model-card: 50 steps; the guidance value feeds the call's
# distilled_guidance_scale (default 3.25), while real CFG runs inside the repo guiders.
("hunyuanimage", 50, 3.25),
# Ideogram 4 model-card: 48 steps, guidance 7 (its schedule tapers the last 3 steps to 3.0;
# the loader keeps that taper when the request matches these defaults exactly).
("ideogram", 48, 7.0),

View file

@ -144,6 +144,64 @@ def test_lumina2_bf16_component_table_present():
assert vae_gb <= 0.5
# ── hunyuanimage-2.1 family ──────────────────────────────────────────────────
@pytest.mark.parametrize(
"repo_id",
[
"hunyuanvideo-community/HunyuanImage-2.1-Diffusers",
"QuantStack/HunyuanImage-2.1-GGUF",
# A local GGUF pick where the family keyword lives in the filename (QuantStack's
# actual naming drops the dash: the hunyuanimage2.1 alias covers it).
"QuantStack/HunyuanImage-2.1-GGUF/HunyuanImage2.1-Q4_K_M.gguf",
],
)
def test_detect_family_hunyuanimage21_repos(repo_id):
fam = detect_family(repo_id)
assert fam is not None and fam.name == "hunyuanimage-2.1"
assert fam.pipeline_class == "HunyuanImagePipeline"
assert fam.transformer_class == "HunyuanImageTransformer2DModel"
assert fam.base_repo == "hunyuanvideo-community/HunyuanImage-2.1-Diffusers"
# The call's guidance knob is distilled_guidance_scale; there is no guidance_scale kwarg.
assert fam.cfg_kwarg == "distilled_guidance_scale"
# Published bf16-only upstream; the fp16 fallback stays off.
assert fam.fp16_incompatible is True
def test_detect_family_hunyuanimage21_override_and_30_still_excluded():
assert detect_family("x", override = "hunyuanimage-2.1").name == "hunyuanimage-2.1"
assert detect_family("x", override = "hunyuanimage2.1").name == "hunyuanimage-2.1"
# The HunyuanImage-3.0 structured exclusion must survive the 2.1 family: 3.0 has
# no diffusers pipeline and must stay unknown with its stated reason.
assert detect_family("tencent/HunyuanImage-3.0") is None
assert excluded_model_reason("tencent/HunyuanImage-3.0") is not None
assert excluded_model_reason("hunyuanvideo-community/HunyuanImage-2.1-Diffusers") is None
def test_hunyuanimage21_is_trusted_non_gguf():
# The mirror pipeline loads via from_pretrained -> needs the allowlist.
assert _is_trusted_diffusion_repo("hunyuanvideo-community/HunyuanImage-2.1-Diffusers")
assert not _is_trusted_diffusion_repo("hunyuanvideo-community/some-future-repo")
def test_hunyuanimage21_generation_defaults():
# Card recipe: 50 steps; guidance feeds the call's distilled_guidance_scale (3.25
# default), while classifier-free guidance runs inside the repo's guider components.
assert default_generation_params(
"hunyuanvideo-community/HunyuanImage-2.1-Diffusers"
) == (50, 3.25)
def test_hunyuanimage21_bf16_component_table_present():
fam = detect_family("hunyuanvideo-community/HunyuanImage-2.1-Diffusers")
sizes = family_bf16_components_gb(fam)
assert sizes is not None
transformer_gb, encoders_gb, vae_gb = sizes
# 17B DiT (32.5 GB bf16 on disk) + Qwen2.5-VL 15.5 GB + ByT5 0.8 GB.
assert 30.0 <= transformer_gb <= 35.0
assert 15.0 <= encoders_gb <= 18.0
assert vae_gb <= 1.0
def test_ideogram4_generation_defaults():
# Model-card settings: 48 steps, guidance 7 (the backend keeps the pipeline's
# recommended tapered schedule when the request matches exactly).