diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 4b17f495ea..cce95fc34c 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -2591,6 +2591,105 @@ class LlamaCppBackend: # ── Lifecycle ───────────────────────────────────────────────── + # GGUF ``general.architecture`` values for diffusion / image models. + # llama.cpp proper has no such architectures, so loading one as a chat + # model dies with "unknown model architecture: ''". These match + # the patched stable-diffusion.cpp / ComfyUI-GGUF enums (LLM_ARCH_FLUX, + # LLM_ARCH_QWEN_IMAGE, ...). Unsloth publishes FLUX and Qwen-Image GGUFs + # under https://huggingface.co/collections/unsloth/unsloth-diffusion-ggufs. + # Matched exactly (not as a substring) so a chat arch merely containing a + # short token like "wan"/"sd1" (e.g. "taiwan") is not misrouted to Images. + _DIFFUSION_ARCHES = frozenset( + ( + "qwen_image", + "flux", + "sd1", + "sdxl", + "sd3", + "aura", + "hidream", + "cosmos", + "ltxv", + "hyvid", + "wan", + "lumina2", + ) + ) + + @staticmethod + def _classify_llama_start_failure( + output: str, + gguf_path: Optional[str], + model_identifier: Optional[str], + ) -> str: + """Explain *why* llama-server failed to start, from its output. + + Several distinct failures all otherwise collapse into the same + opaque "invalid GGUF or out of memory" message. The worst case is + a diffusion / image GGUF (FLUX, Qwen-Image, ...) loaded as a chat + model: the file is perfectly valid and there is plenty of memory, + but llama.cpp has no such architecture, so the user is told to free + memory that was never the problem (issue #5842). Pick the most + specific message the captured output supports. + """ + lowered = (output or "").lower() + + # Detect Ollama source up front so the arch branch can keep the + # Ollama hint instead of the generic "unsupported arch" message. + gguf = gguf_path or "" + is_ollama = ( + ".studio_links" in gguf + or os.sep + "ollama_links" + os.sep in gguf + or os.sep + ".cache" + os.sep + "ollama" + os.sep in gguf + or (model_identifier or "").startswith("ollama/") + ) + + # "unknown model architecture: ''": diffusion -> Images page, + # Ollama -> Ollama hint, else a precise "unsupported" message. Exact + # match so chat archs are never misrouted. + arch_match = re.search(r"unknown model architecture:\s*'([^']+)'", lowered) + if arch_match: + arch = arch_match.group(1) + if arch in LlamaCppBackend._DIFFUSION_ARCHES: + return ( + f"'{arch}' is a diffusion (image-generation) GGUF, which " + "llama-server cannot run as a chat/completion model. Use " + "Studio's Images page to generate with local diffusion " + "GGUFs such as FLUX and Qwen-Image." + ) + if is_ollama: + return ( + "Some Ollama models do not work with llama.cpp. Try a " + "different model, or use this model directly through " + "Ollama instead." + ) + return ( + f"llama.cpp does not support this GGUF's model architecture " + f"('{arch}'). The file is valid, but this model type cannot " + "be run with llama-server." + ) + + # Other Ollama compat failures that do not name an arch. Only when + # the output shows a GGUF compat issue, not OOM / missing binaries. + if is_ollama: + gguf_compat_hints = ( + "key not found", + "unknown model architecture", + "failed to load model", + ) + if any(h in lowered for h in gguf_compat_hints): + return ( + "Some Ollama models do not work with llama.cpp. Try a " + "different model, or use this model directly through " + "Ollama instead." + ) + + # Fallback: genuinely unknown failure (OOM, missing binary, ...). + return ( + "llama-server failed to start. " + "Check that the GGUF file is valid and you have enough memory." + ) + def load_model( self, *, @@ -3383,31 +3482,12 @@ class LlamaCppBackend: # Wait for llama-server to become healthy if not self._wait_for_health(timeout = 600.0): self._kill_process() - _gguf = gguf_path or "" - _is_ollama = ( - ".studio_links" in _gguf - or os.sep + "ollama_links" + os.sep in _gguf - or os.sep + ".cache" + os.sep + "ollama" + os.sep in _gguf - or (self._model_identifier or "").startswith("ollama/") - ) - # Only show the Ollama-specific message when the server - # output indicates a GGUF compatibility issue, not for - # unrelated failures like OOM or missing binaries. - if _is_ollama: - _output = "\n".join(self._stdout_lines[-50:]).lower() - _gguf_compat_hints = ( - "key not found", - "unknown model architecture", - "failed to load model", - ) - if any(h in _output for h in _gguf_compat_hints): - raise RuntimeError( - "Some Ollama models do not work with llama.cpp. " - "Try a different model, or use this model directly through Ollama instead." - ) raise RuntimeError( - "llama-server failed to start. " - "Check that the GGUF file is valid and you have enough memory." + self._classify_llama_start_failure( + "\n".join(self._stdout_lines[-50:]), + gguf_path, + self._model_identifier, + ) ) self._healthy = True diff --git a/studio/backend/tests/test_llama_cpp_start_failure_classification.py b/studio/backend/tests/test_llama_cpp_start_failure_classification.py new file mode 100644 index 0000000000..e647ff2c7d --- /dev/null +++ b/studio/backend/tests/test_llama_cpp_start_failure_classification.py @@ -0,0 +1,144 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0 + +"""Tests for LlamaCppBackend._classify_llama_start_failure. + +When llama-server exits before becoming healthy, load_model turns its +captured stdout/stderr into a user-facing reason. A diffusion / image +GGUF (FLUX, Qwen-Image, ...) is a valid file with plenty of memory, so +the generic "invalid file or out of memory" message is actively +misleading (issue #5842). These tests pin the classification. +""" + +from __future__ import annotations + +import sys +import types as _types +from pathlib import Path + +import pytest + +_BACKEND_DIR = str(Path(__file__).resolve().parent.parent) +if _BACKEND_DIR not in sys.path: + sys.path.insert(0, _BACKEND_DIR) + +# Match the stubbing pattern in sibling tests so the module imports in a +# lightweight env without fastapi. +_loggers_stub = _types.ModuleType("loggers") +_loggers_stub.get_logger = lambda name: __import__("logging").getLogger(name) +sys.modules.setdefault("loggers", _loggers_stub) +# Give the structlog stub a real get_logger: a bare ModuleType poisons +# sys.modules for later tests that call structlog.get_logger at import time. +_structlog_stub = _types.ModuleType("structlog") +_structlog_stub.get_logger = lambda *a, **k: __import__("logging").getLogger( + "structlog" +) +sys.modules.setdefault("structlog", _structlog_stub) +if not hasattr(sys.modules["structlog"], "get_logger"): + sys.modules["structlog"].get_logger = _structlog_stub.get_logger + +from core.inference.llama_cpp import LlamaCppBackend # noqa: E402 + +_classify = LlamaCppBackend._classify_llama_start_failure + +# Real llama-server failure lines (lower-cased downstream anyway). +_QWEN_IMAGE_OUT = ( + "load_model: loading model 'qwen-image-edit-2511-Q4_K_M.gguf'\n" + "llama_model_load: error loading model: unknown model architecture: 'qwen_image'\n" + "llama_model_load_from_file_impl: failed to load model" +) +_OOM_OUT = ( + "ggml_backend_cuda_buffer_type_alloc_buffer: allocating 12000.00 MiB on " + "device 0: cudaMalloc failed: out of memory" +) + + +class TestDiffusionArchitectures: + def test_qwen_image_routes_to_images_page(self): + msg = _classify(_QWEN_IMAGE_OUT, "/models/qwen-image.gguf", "local/qwen-image") + assert "diffusion" in msg.lower() + assert "Images page" in msg + assert "qwen_image" in msg + # Must NOT keep blaming memory / file validity. + assert "out of memory" not in msg.lower() + assert "enough memory" not in msg.lower() + + # Parametrize over the production set so new arches are auto-covered. + @pytest.mark.parametrize("arch", sorted(LlamaCppBackend._DIFFUSION_ARCHES)) + def test_every_diffusion_arch_is_recognised(self, arch): + out = f"error loading model: unknown model architecture: '{arch}'" + msg = _classify(out, f"/models/{arch}.gguf", f"local/{arch}") + assert "diffusion" in msg.lower() + assert "Images page" in msg + assert arch in msg + + +class TestUnsupportedNonDiffusionArchitecture: + def test_unknown_llm_arch_says_unsupported_not_oom(self): + out = "error loading model: unknown model architecture: 'some_new_llm'" + msg = _classify(out, "/models/x.gguf", "local/x") + assert "some_new_llm" in msg + assert "architecture" in msg.lower() + # Specific, not the misleading memory message. + assert "enough memory" not in msg.lower() + assert "diffusion" not in msg.lower() + + # Exact match: a chat arch merely containing a diffusion token (wan, + # sd1, flux, ...) must not be routed to the Images page. + @pytest.mark.parametrize( + "arch", + [ + "taiwan", # contains "wan" + "swan_llm", # contains "wan" + "fluxion", # contains "flux" + "sd1234", # contains "sd1" + "sd3_chat", # contains "sd3" + "aura2_text", # contains "aura" + "cosmos_reason", # contains "cosmos" + "qwen_image_text", # contains "qwen_image" + ], + ) + def test_arch_containing_diffusion_token_is_not_misrouted(self, arch): + out = f"error loading model: unknown model architecture: '{arch}'" + msg = _classify(out, f"/models/{arch}.gguf", f"local/{arch}") + assert arch in msg + assert "does not support" in msg.lower() + assert "diffusion" not in msg.lower() + assert "Images page" not in msg + + +class TestOllamaAndFallback: + _OLLAMA_GGUF = ( + f"/home/u/.ollama{__import__('os').sep}ollama_links" + f"{__import__('os').sep}m.gguf" + ) + + def test_ollama_compat_message_still_works(self): + out = "llama_model_load: error loading model: key not found" + msg = _classify(out, self._OLLAMA_GGUF, "ollama/llama3") + assert "Ollama" in msg + + def test_ollama_unknown_arch_keeps_ollama_guidance(self): + # Ollama + non-diffusion unknown arch keeps the Ollama hint, not the + # generic llama.cpp "unsupported" message. + out = "error loading model: unknown model architecture: 'some_new_llm'" + msg = _classify(out, self._OLLAMA_GGUF, "ollama/some-new") + assert "Ollama" in msg + assert "directly through Ollama" in msg + assert "does not support" not in msg.lower() + + def test_ollama_diffusion_arch_still_routes_to_images(self): + # Diffusion routing wins over the Ollama hint. + out = "error loading model: unknown model architecture: 'flux'" + msg = _classify(out, self._OLLAMA_GGUF, "ollama/flux") + assert "diffusion" in msg.lower() + assert "Images page" in msg + + def test_generic_oom_keeps_memory_message(self): + msg = _classify(_OOM_OUT, "/models/big.gguf", "local/big") + assert "enough memory" in msg.lower() + assert "diffusion" not in msg.lower() + + def test_empty_output_is_safe(self): + msg = _classify("", None, None) + assert "llama-server failed to start" in msg