diff --git a/studio/backend/core/inference/diffusion.py b/studio/backend/core/inference/diffusion.py index dee46a3704..ea22828d68 100644 --- a/studio/backend/core/inference/diffusion.py +++ b/studio/backend/core/inference/diffusion.py @@ -73,11 +73,18 @@ class DiffusionFamily: _FAMILIES: tuple[DiffusionFamily, ...] = ( + # The "9b" alias is checked first so a "flux-2-klein-9b" GGUF picks + # the 9B base instead of the 4B one when the user does not pass an + # explicit base_repo. Apache 2.0 is preferred as the auto-default for + # the 4B path because BFL's 9B base is gated. DiffusionFamily( name = "flux.2-klein", pipeline_class = "Flux2KleinPipeline", transformer_class = "Flux2Transformer2DModel", - base_repo = "black-forest-labs/FLUX.2-klein", + # Default for klein when no explicit base_repo: Apache-2.0 4B Base. + # The frontend curated picker always passes base_repo explicitly, + # so this default only fires for "custom HF repo" mode. + base_repo = "black-forest-labs/FLUX.2-klein-base-4B", aliases = ("flux2-klein", "flux-2-klein", "flux.2.klein"), ), DiffusionFamily( @@ -111,7 +118,13 @@ _FAMILIES: tuple[DiffusionFamily, ...] = ( DiffusionFamily( name = "stable-diffusion-xl", pipeline_class = "StableDiffusionXLPipeline", - transformer_class = "", # SDXL uses a UNet, not a transformer + # SDXL uses a UNet, not a transformer. Loading SDXL GGUFs would + # require UNet2DConditionModel.from_single_file + GGUF, which is + # not the same code path as the FLUX / Qwen-Image transformers + # this PR ships. Until that path is wired and smoke-tested, + # treat SDXL as full-repo-only and surface a clear error when a + # user tries to pass gguf_filename for it. + transformer_class = "", base_repo = "stabilityai/stable-diffusion-xl-base-1.0", aliases = ("sdxl",), ), @@ -171,7 +184,15 @@ class DiffusionBackend: def __init__(self) -> None: self._pipe: Any = None + # `_lock` protects mutations to the small state fields and the + # pipe call inside generate_image. `_load_lock` serialises the + # entire load_model call so two concurrent /images/load requests + # cannot both reach pipeline_cls.from_pretrained at the same + # time (which would double-spend VRAM and corrupt _pipe). The + # locks are taken in order load -> state so a generation in + # flight cannot deadlock the next load. self._lock = threading.Lock() + self._load_lock = threading.Lock() self._family: Optional[DiffusionFamily] = None self._repo_id: Optional[str] = None self._gguf_path: Optional[str] = None @@ -269,86 +290,109 @@ class DiffusionBackend: device, dtype = self._pick_device_and_dtype() - with self._lock: - self._loading = True - self._last_error = None - try: - pipeline_cls = getattr(diffusers, fam.pipeline_class, None) - if pipeline_cls is None: - raise RuntimeError( - f"diffusers {diffusers.__version__} has no " - f"{fam.pipeline_class}; upgrade diffusers and retry." - ) - transformer_cls = ( - getattr(diffusers, fam.transformer_class, None) - if fam.transformer_class - else None - ) + # _load_lock serialises the entire load so two concurrent calls + # cannot both kick off a multi-GB download + GPU upload at once. + # The second caller waits behind the first and then loads on top + # of the now-populated state via the normal swap path. + with self._load_lock: + with self._lock: + self._loading = True + self._last_error = None + try: + # Unload any chat model that is holding GPU memory so the + # diffusion load does not OOM on a < 24 GB GPU. Best + # effort: if the llama-cpp backend module is absent (eg + # tests, headless tooling) we just continue. + _release_chat_backend_for_diffusion() - effective_base = base_repo or fam.base_repo - logger.info( - "Loading diffusion model %s (family=%s, device=%s, dtype=%s, base=%s)", - repo_id, - fam.name, - device, - dtype, - effective_base, - ) - - transformer = None - local_gguf_path: Optional[str] = None - if gguf_filename: - if transformer_cls is None: + pipeline_cls = getattr(diffusers, fam.pipeline_class, None) + if pipeline_cls is None: raise RuntimeError( - f"Family {fam.name} does not have a GGUF transformer " - "path; load the full repo instead." + f"diffusers {diffusers.__version__} has no " + f"{fam.pipeline_class}; upgrade diffusers and retry." ) - local_gguf_path = hf_hub_download( - repo_id = repo_id, - filename = gguf_filename, - token = hf_token, - ) - quant_config = diffusers.GGUFQuantizationConfig(compute_dtype = dtype) - transformer = transformer_cls.from_single_file( - local_gguf_path, - quantization_config = quant_config, - torch_dtype = dtype, + transformer_cls = ( + getattr(diffusers, fam.transformer_class, None) + if fam.transformer_class + else None ) - pipe_kwargs: dict[str, Any] = {"torch_dtype": dtype} - if transformer is not None: - pipe_kwargs["transformer"] = transformer - if hf_token: - pipe_kwargs["token"] = hf_token + # Resolution rules for the "what repo to call + # from_pretrained on" question: + # 1. caller-supplied base_repo wins + # 2. if no GGUF file was requested the user is loading a + # full diffusers repo; use repo_id directly so we do + # not silently substitute the family default + # 3. otherwise fall back to the family default + if base_repo: + effective_base = base_repo + elif not gguf_filename: + effective_base = repo_id + else: + effective_base = fam.base_repo + logger.info( + "Loading diffusion model %s (family=%s, device=%s, dtype=%s, base=%s)", + repo_id, + fam.name, + device, + dtype, + effective_base, + ) - pipe = pipeline_cls.from_pretrained(effective_base, **pipe_kwargs) - if enable_model_cpu_offload and device == "cuda": - pipe.enable_model_cpu_offload() - else: - pipe.to(device) + transformer = None + local_gguf_path: Optional[str] = None + if gguf_filename: + if transformer_cls is None: + raise RuntimeError( + f"Family {fam.name} does not have a GGUF transformer " + "path wired in this build; load the full repo instead." + ) + local_gguf_path = hf_hub_download( + repo_id = repo_id, + filename = gguf_filename, + token = hf_token, + ) + quant_config = diffusers.GGUFQuantizationConfig(compute_dtype = dtype) + transformer = transformer_cls.from_single_file( + local_gguf_path, + quantization_config = quant_config, + torch_dtype = dtype, + ) - # Drop the old pipeline only after the new one is in place. - old = self._pipe - with self._lock: - self._pipe = pipe - self._family = fam - self._repo_id = repo_id - self._gguf_path = local_gguf_path - self._base_repo = effective_base - self._device = device - self._dtype = str(dtype).replace("torch.", "") - self._loaded_at = time.time() - _release(old) + pipe_kwargs: dict[str, Any] = {"torch_dtype": dtype} + if transformer is not None: + pipe_kwargs["transformer"] = transformer + if hf_token: + pipe_kwargs["token"] = hf_token - return self.status() - except Exception as exc: - with self._lock: - self._last_error = str(exc) - logger.exception("Diffusion load failed for %s", repo_id) - raise RuntimeError(f"Failed to load diffusion model: {exc}") from exc - finally: - with self._lock: - self._loading = False + pipe = pipeline_cls.from_pretrained(effective_base, **pipe_kwargs) + if enable_model_cpu_offload and device == "cuda": + pipe.enable_model_cpu_offload() + else: + pipe.to(device) + + # Drop the old pipeline only after the new one is in place. + old = self._pipe + with self._lock: + self._pipe = pipe + self._family = fam + self._repo_id = repo_id + self._gguf_path = local_gguf_path + self._base_repo = effective_base + self._device = device + self._dtype = str(dtype).replace("torch.", "") + self._loaded_at = time.time() + _release(old) + + return self.status() + except Exception as exc: + with self._lock: + self._last_error = str(exc) + logger.exception("Diffusion load failed for %s", repo_id) + raise RuntimeError(f"Failed to load diffusion model: {exc}") from exc + finally: + with self._lock: + self._loading = False def unload_model(self) -> dict[str, Any]: with self._lock: @@ -420,8 +464,18 @@ class DiffusionBackend: "width": int(width), "height": int(height), } + # FLUX.2 / FLUX.2 klein pipelines do NOT accept + # negative_prompt and 500 if you pass it in. Inspect the + # signature and only forward when supported; warn otherwise + # so the UI can disable the field for incompatible families. if negative_prompt is not None and negative_prompt.strip(): - call_kwargs["negative_prompt"] = negative_prompt + if _pipe_accepts_kwarg(pipe, "negative_prompt"): + call_kwargs["negative_prompt"] = negative_prompt + else: + logger.info( + "Dropping negative_prompt: %s does not accept it", + type(pipe).__name__, + ) if generator is not None: call_kwargs["generator"] = generator @@ -432,6 +486,26 @@ class DiffusionBackend: return images[0] +def _pipe_accepts_kwarg(pipe: Any, name: str) -> bool: + """True if ``pipe.__call__`` advertises a kwarg called ``name``. + + Cheap inspect-based probe so we do not have to maintain a manual + list of which pipeline classes accept negative_prompt. Returns + False on any introspection error so callers stay on the safe path. + """ + import inspect + + try: + sig = inspect.signature(pipe.__call__) + except (TypeError, ValueError): + return False + if name in sig.parameters: + return True + return any( + p.kind is inspect.Parameter.VAR_KEYWORD for p in sig.parameters.values() + ) + + def encode_png_base64(pil_image: "Any") -> str: """Encode a PIL image to base64-encoded PNG.""" import base64 @@ -444,6 +518,36 @@ def encode_png_base64(pil_image: "Any") -> str: # ─── Helpers ────────────────────────────────────────────────────────── +def _release_chat_backend_for_diffusion() -> None: + """Unload any running chat backend before a diffusion load. + + Diffusion pipelines on FLUX-class models can eat 12-24 GB of VRAM, + and llama-server typically holds onto its loaded GGUF until told to + drop it. Asking the chat backend to release its weights first means + a typical 24 GB consumer GPU can host one chat model OR one + diffusion model without manual unload steps. + + Best effort: if the chat backend module is not importable (CI, + isolated tests, custom builds) we silently continue. Failures + inside the unload itself are logged but not propagated; the + diffusion load can still try and surface its own OOM. + """ + try: + from routes.inference import get_llama_cpp_backend # type: ignore + except Exception: + return + try: + backend = get_llama_cpp_backend() + except Exception: + return + try: + if getattr(backend, "is_loaded", False): + logger.info("Unloading llama-server before diffusion load") + backend.unload_model() + except Exception as exc: + logger.warning("Could not unload chat backend before diffusion: %s", exc) + + def _release(obj: Any) -> None: """Best-effort GPU-memory release for a pipeline being swapped out.""" if obj is None: diff --git a/studio/backend/tests/test_diffusion_backend.py b/studio/backend/tests/test_diffusion_backend.py index 642af72361..67deed9c5e 100644 --- a/studio/backend/tests/test_diffusion_backend.py +++ b/studio/backend/tests/test_diffusion_backend.py @@ -53,6 +53,11 @@ def test_detect_family_flux2_klein(): assert fam.name == "flux.2-klein" assert fam.pipeline_class == "Flux2KleinPipeline" assert fam.transformer_class == "Flux2Transformer2DModel" + # Family default base must point to a real Hub repo (not the bare + # "FLUX.2-klein" slug that does not exist). The frontend curated + # picker still passes base_repo explicitly per size so this default + # only fires for the "custom HF repo" mode. + assert fam.base_repo == "black-forest-labs/FLUX.2-klein-base-4B" def test_detect_family_flux2_dev_is_not_klein(): @@ -363,14 +368,14 @@ def test_load_model_gguf_path_happy(monkeypatch): backend = get_diffusion_backend() status = backend.load_model( "unsloth/FLUX.2-klein-4B-GGUF", - gguf_filename = "FLUX.2-klein-4B-Q4_K_S.gguf", + gguf_filename = "flux-2-klein-4b-Q4_K_S.gguf", ) assert status["is_loaded"] is True assert status["family"] == "flux.2-klein" assert status["pipeline_class"] == "Flux2KleinPipeline" - assert status["base_repo"] == "black-forest-labs/FLUX.2-klein" + assert status["base_repo"] == "black-forest-labs/FLUX.2-klein-base-4B" assert status["gguf_path"] == ( - "/fake/unsloth/FLUX.2-klein-4B-GGUF/FLUX.2-klein-4B-Q4_K_S.gguf" + "/fake/unsloth/FLUX.2-klein-4B-GGUF/flux-2-klein-4b-Q4_K_S.gguf" ) @@ -397,12 +402,223 @@ def test_load_model_swap_drops_previous(monkeypatch): backend = get_diffusion_backend() backend.load_model( "unsloth/FLUX.2-klein-4B-GGUF", - gguf_filename = "FLUX.2-klein-4B-Q4_K_S.gguf", + gguf_filename = "flux-2-klein-4b-Q4_K_S.gguf", ) first_pipe = backend._pipe backend.load_model( "unsloth/FLUX.2-dev-GGUF", - gguf_filename = "FLUX.2-dev-Q4_K_S.gguf", + gguf_filename = "flux2-dev-Q4_K_S.gguf", ) assert backend._pipe is not first_pipe assert backend.status()["family"] == "flux.2" + + +def test_load_model_base_repo_override(monkeypatch): + _install_fake_diffusers(monkeypatch) + from core.inference.diffusion import get_diffusion_backend + + backend = get_diffusion_backend() + status = backend.load_model( + "unsloth/FLUX.2-klein-9B-GGUF", + gguf_filename = "flux-2-klein-9b-Q4_K_S.gguf", + base_repo = "black-forest-labs/FLUX.2-klein-base-9B", + ) + assert status["base_repo"] == "black-forest-labs/FLUX.2-klein-base-9B" + + +def test_load_model_full_repo_does_not_substitute(monkeypatch): + """A full diffusers repo (no gguf_filename) must call from_pretrained + with the user-supplied repo, not the family default. This was the + silent-substitution bug surfaced by review.""" + fake = _install_fake_diffusers(monkeypatch) + from core.inference.diffusion import get_diffusion_backend + + backend = get_diffusion_backend() + status = backend.load_model( + "owner/FLUX.1-finetune-diffusers", + family_override = "flux.1", + ) + # base_repo must echo the user repo, not the family default. + assert status["base_repo"] == "owner/FLUX.1-finetune-diffusers" + assert status["repo_id"] == "owner/FLUX.1-finetune-diffusers" + # And the fake pipeline records what we called from_pretrained with. + assert backend._pipe.base_repo == "owner/FLUX.1-finetune-diffusers" + + +def test_load_model_concurrent_serialises(monkeypatch): + """Two concurrent load_model() calls must NOT both reach + pipeline_cls.from_pretrained at the same time (race fix).""" + _install_fake_diffusers(monkeypatch) + from core.inference.diffusion import get_diffusion_backend + import threading + import time as _t + + backend = get_diffusion_backend() + active = {"n": 0, "max": 0} + lock = threading.Lock() + + import sys as _sys + + fake_pipeline_cls = _sys.modules["diffusers"].Flux2KleinPipeline + original_from_pretrained = fake_pipeline_cls.from_pretrained.__func__ + + def _instrumented_from_pretrained(cls, base_repo, **kwargs): + with lock: + active["n"] += 1 + active["max"] = max(active["max"], active["n"]) + try: + _t.sleep(0.1) + return original_from_pretrained(cls, base_repo, **kwargs) + finally: + with lock: + active["n"] -= 1 + + fake_pipeline_cls.from_pretrained = classmethod(_instrumented_from_pretrained) + + errors: list = [] + + def _do_load(): + try: + backend.load_model( + "unsloth/FLUX.2-klein-base-4B-GGUF", + gguf_filename = "flux-2-klein-base-4b-Q4_K_S.gguf", + ) + except Exception as e: + errors.append(e) + + threads = [threading.Thread(target = _do_load) for _ in range(3)] + for t in threads: + t.start() + for t in threads: + t.join() + + assert not errors, errors + assert active["max"] == 1, ( + f"Expected concurrent loads to serialise; max_active={active['max']}" + ) + + +def test_pipe_accepts_kwarg_filter(): + """The negative_prompt filter must drop the kwarg on classes that + do not accept it (FLUX.2 / FLUX.2 klein) and keep it on the rest.""" + from core.inference.diffusion import _pipe_accepts_kwarg + + class _NoNeg: + def __call__(self, *, prompt, num_inference_steps, guidance_scale, width, height): + pass + + class _Neg: + def __call__( + self, + *, + prompt, + negative_prompt = None, + num_inference_steps, + guidance_scale, + width, + height, + ): + pass + + class _VarKw: + def __call__(self, **kw): + pass + + assert _pipe_accepts_kwarg(_NoNeg(), "negative_prompt") is False + assert _pipe_accepts_kwarg(_Neg(), "negative_prompt") is True + # Anything with **kwargs is assumed to accept the kwarg (the + # alternative is to silently drop legitimate params). + assert _pipe_accepts_kwarg(_VarKw(), "negative_prompt") is True + + +def test_generate_image_strips_negative_prompt_on_flux2(monkeypatch): + """generate_image must drop negative_prompt when the loaded pipeline + does not accept it; otherwise FLUX.2 would 500 on a user-visible + field.""" + import core.inference.diffusion as d + from PIL import Image + + backend = d.get_diffusion_backend() + + received: dict = {} + + class _Flux2LikePipe: + # Signature mirrors Flux2Pipeline.__call__: NO negative_prompt. + # No **kw either, since the real FLUX.2 pipeline does not accept + # arbitrary kwargs (passing negative_prompt to it raises TypeError). + def __call__( + self, + *, + prompt, + num_inference_steps, + guidance_scale, + width, + height, + generator = None, + ): + received["prompt"] = prompt + class _Out: + pass + o = _Out() + o.images = [Image.new("RGB", (width, height), (1, 2, 3))] + return o + + backend._pipe = _Flux2LikePipe() + backend._device = "cpu" + backend._family = d._FAMILIES[0] + backend._repo_id = "stub/stub" + + # If generate_image forwarded negative_prompt, the pipeline call + # would raise TypeError. The PR's filter drops it, so the call + # succeeds and we observe the prompt was still delivered. + backend.generate_image( + prompt = "a sloth", + negative_prompt = "blurry, low quality", + num_inference_steps = 4, + guidance_scale = 1.0, + width = 256, + height = 256, + ) + assert received["prompt"] == "a sloth" + + +def test_generate_image_keeps_negative_prompt_on_supporting_pipe(monkeypatch): + import core.inference.diffusion as d + from PIL import Image + + backend = d.get_diffusion_backend() + captured: dict = {} + + class _NegOK: + def __call__( + self, + *, + prompt, + negative_prompt = None, + num_inference_steps, + guidance_scale, + width, + height, + **kw, + ): + captured["negative_prompt"] = negative_prompt + class _Out: + pass + o = _Out() + o.images = [Image.new("RGB", (width, height), (4, 5, 6))] + return o + + backend._pipe = _NegOK() + backend._device = "cpu" + backend._family = d._FAMILIES[2] # flux.1 supports negative_prompt + backend._repo_id = "stub/stub" + + backend.generate_image( + prompt = "a sloth", + negative_prompt = "blurry", + num_inference_steps = 4, + guidance_scale = 1.0, + width = 256, + height = 256, + ) + assert captured["negative_prompt"] == "blurry" diff --git a/studio/backend/tests/test_diffusion_routes.py b/studio/backend/tests/test_diffusion_routes.py index 9b9063f0b1..af759ab234 100644 --- a/studio/backend/tests/test_diffusion_routes.py +++ b/studio/backend/tests/test_diffusion_routes.py @@ -118,7 +118,7 @@ def test_load_then_generate_round_trip(app_with_stub): "/api/inference/images/load", json = { "repo_id": "unsloth/FLUX.2-klein-4B-GGUF", - "gguf_filename": "FLUX.2-klein-4B-Q4_K_S.gguf", + "gguf_filename": "flux-2-klein-4b-Q4_K_S.gguf", }, ) assert r.status_code == 200, r.text diff --git a/studio/frontend/src/features/images/images-page.tsx b/studio/frontend/src/features/images/images-page.tsx index 8928413313..a422b59bbc 100644 --- a/studio/frontend/src/features/images/images-page.tsx +++ b/studio/frontend/src/features/images/images-page.tsx @@ -28,47 +28,64 @@ import { } from "./api"; import { useCallback, useEffect, useMemo, useRef, useState } from "react"; -// Curated short list of working unsloth/* diffusion GGUFs. Picked to -// span size + license so any GPU class has at least one viable option: -// FLUX.2 klein 4B -> ~10-12 GB VRAM with Q4_K_S, Apache 2.0 -// FLUX.2 klein 9B -> ~16-18 GB VRAM, FLUX [klein] non-commercial -// FLUX.2 dev -> ~24+ GB VRAM, FLUX [dev] non-commercial -// The CLI on the backend can load anything supported by detect_family(); -// this list just keeps the picker compact for the v1 UI. +// Curated short list of working diffusion GGUFs. Picked to span +// size + license so any GPU class has at least one viable option: +// FLUX.2 klein 4B -> ~13 GB VRAM with Q4_K_S, Apache 2.0 +// FLUX.2 klein 9B -> ~17 GB VRAM, FLUX [klein] non-commercial (gated) +// FLUX.2 dev -> ~24+ GB VRAM, FLUX [dev] non-commercial (gated) +// FLUX.1 dev -> ~12 GB VRAM, older but widely tested (gated) +// +// Filenames mirror the Hub canonical case (lowercase 'flux-2-klein-4b') +// and base_repo is set explicitly so the backend never falls back to the +// family default. The CLI on the backend can load anything supported by +// detect_family(); this list just keeps the picker compact for the v1 UI. const CURATED_MODELS: Array<{ label: string; repo_id: string; default_gguf: string; + base_repo: string; family: string; notes: string; }> = [ { - label: "FLUX.2 klein 4B (Q4_K_S, Apache 2.0)", + label: "FLUX.2 klein base 4B (Q4_K_S, Apache 2.0)", + repo_id: "unsloth/FLUX.2-klein-base-4B-GGUF", + default_gguf: "flux-2-klein-base-4b-Q4_K_S.gguf", + base_repo: "black-forest-labs/FLUX.2-klein-base-4B", + family: "flux.2-klein", + notes: "13 GB VRAM, fastest. Apache 2.0, ungated.", + }, + { + label: "FLUX.2 klein 4B (Q4_K_S, distilled)", repo_id: "unsloth/FLUX.2-klein-4B-GGUF", - default_gguf: "FLUX.2-klein-4B-Q4_K_S.gguf", + default_gguf: "flux-2-klein-4b-Q4_K_S.gguf", + base_repo: "black-forest-labs/FLUX.2-klein-base-4B", family: "flux.2-klein", - notes: "13 GB VRAM, fastest. Apache 2.0.", + notes: "13 GB VRAM. Distilled klein 4B with the Apache base.", }, { - label: "FLUX.2 klein 9B (Q4_K_S)", + label: "FLUX.2 klein 9B (Q4_K_S, gated)", repo_id: "unsloth/FLUX.2-klein-9B-GGUF", - default_gguf: "FLUX.2-klein-9B-Q4_K_S.gguf", + default_gguf: "flux-2-klein-9b-Q4_K_S.gguf", + base_repo: "black-forest-labs/FLUX.2-klein-base-9B", family: "flux.2-klein", - notes: "17 GB VRAM, higher quality.", + notes: "17 GB VRAM. Higher quality. Requires HF access to FLUX.2 klein base 9B.", }, { - label: "FLUX.2 dev (Q4_K_S)", + label: "FLUX.2 dev (Q4_K_S, gated)", repo_id: "unsloth/FLUX.2-dev-GGUF", - default_gguf: "FLUX.2-dev-Q4_K_S.gguf", + default_gguf: "flux2-dev-Q4_K_S.gguf", + base_repo: "black-forest-labs/FLUX.2-dev", family: "flux.2", - notes: "24+ GB VRAM, best for prompt following.", + notes: "24+ GB VRAM. Requires HF access to FLUX.2 dev.", }, { - label: "FLUX.1 dev (Q4_K_S, city96)", + label: "FLUX.1 dev (Q4_K_S, city96, gated)", repo_id: "city96/FLUX.1-dev-gguf", default_gguf: "flux1-dev-Q4_K_S.gguf", + base_repo: "black-forest-labs/FLUX.1-dev", family: "flux.1", - notes: "12 GB VRAM, older but well tested.", + notes: "12 GB VRAM. Older but widely tested. Requires HF access to FLUX.1 dev.", }, ]; @@ -132,6 +149,11 @@ export function ImagesPage() { const repo = useCustom ? customRepoId.trim() : preset.repo_id; const gguf = useCustom ? customGguf.trim() || undefined : preset.default_gguf; const family = useCustom ? undefined : preset.family; + // Always pass base_repo for curated entries; custom-repo mode + // lets the backend either infer it from the family default or + // (when no GGUF is given) treat the repo as a full diffusers + // checkpoint and call from_pretrained on it directly. + const baseRepo = useCustom ? undefined : preset.base_repo; if (!repo) { toast.error("Pick a model first"); return; @@ -139,6 +161,7 @@ export function ImagesPage() { const next = await loadDiffusionModel({ repo_id: repo, gguf_filename: gguf, + base_repo: baseRepo, family, hf_token: hfToken.trim() || undefined, }); @@ -208,6 +231,19 @@ export function ImagesPage() { return "Not loaded"; }, [status, refreshingStatus]); + // FLUX.2 / FLUX.2 klein pipelines do NOT accept negative_prompt and + // would 500 if we sent one through. The backend strips the field + // defensively but hiding it client-side keeps the UI honest. + const supportsNegativePrompt = useMemo(() => { + const family = status?.family; + if (!family) { + const candidate = useCustom ? undefined : preset.family; + if (!candidate) return true; + return !candidate.startsWith("flux.2"); + } + return !family.startsWith("flux.2"); + }, [status, useCustom, preset.family]); + return (
-
- -