From c8bcacc3fea27e97bea0ea09c9ad7554c729724f Mon Sep 17 00:00:00 2001 From: oobabooga Date: Sat, 27 Jun 2026 02:43:36 -0300 Subject: [PATCH] Fix fast_inference crash on ABI-broken vLLM: probe compiled extensions, not just import vllm (#6621) * Fix fast_inference crash on ABI-broken vLLM: force-load compiled extensions in the broken-vLLM probe * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * Broaden broken-vLLM probe: catch non-libcudart .so failures and _moe_C_stable_libtorch * Revert stray reformat of the PDL fix log line * Trim verbose comments in the broken-vLLM probe * Drop non-existent vllm._moe_C_stable_libtorch from the broken-vLLM probe * Shorten comments in broken vLLM extension detection Condense the docstrings and inline comments for the lazy-loaded vLLM probe and the new regression test while keeping the rationale. Comments only, no code changes (verified with an AST signature check and the existing tests). --------- Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com> Co-authored-by: Daniel Han --- tests/test_vllm_broken_detection.py | 166 ++++++++++++++++++++++++++++ unsloth/_gpu_init.py | 19 ++-- unsloth/import_fixes.py | 27 ++++- 3 files changed, 197 insertions(+), 15 deletions(-) create mode 100644 tests/test_vllm_broken_detection.py diff --git a/tests/test_vllm_broken_detection.py b/tests/test_vllm_broken_detection.py new file mode 100644 index 0000000000..ee89ddbedf --- /dev/null +++ b/tests/test_vllm_broken_detection.py @@ -0,0 +1,166 @@ +# Unsloth - 2x faster, 60% less VRAM LLM training and finetuning +# Copyright 2023-present Daniel Han-Chen, Michael Han-Chen & the Unsloth team. All rights reserved. +# +# This program is free software: you can redistribute it and/or modify +# it under the terms of the GNU Lesser General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# This program is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU Lesser General Public License for more details. + +"""Regression test for #6590: modern vLLM lazy-loads its compiled extensions, so +a bare ``import vllm`` succeeds even when ``vllm._C`` (or a sibling) is ABI-broken +and ``disable_broken_vllm`` missed it. GPU-free, via a synthetic vLLM.""" + +from __future__ import annotations + +import contextlib +import importlib.abc +import importlib.machinery +import importlib.util +import sys +import types + +import pytest + + +_LIBCUDART_ERROR = "libcudart.so.13: cannot open shared object file: No such file or directory" + + +class _ExtensionLoader(importlib.abc.Loader): + """A compiled extension that loads cleanly or fails on dlopen.""" + + def __init__(self, broken, error): + self.broken = broken + self.error = error + + def create_module(self, spec): + return None + + def exec_module(self, module): + if self.broken: + raise ImportError(self.error) + + +class _FakeVllmFinder(importlib.abc.MetaPathFinder): + """Lazy vLLM: ``import vllm`` succeeds; each ``vllm._*`` ext is healthy, + ABI-broken, or absent, as real vLLM only loads ``_C`` & friends on use.""" + + def __init__(self, present, broken, error): + self.present = present + self.broken = broken + self.error = error + + def find_spec( + self, + fullname, + path = None, + target = None, + ): + if fullname in self.present: + return importlib.machinery.ModuleSpec( + name = fullname, + loader = _ExtensionLoader(broken = fullname in self.broken, error = self.error), + is_package = False, + ) + return None # absent -> ModuleNotFoundError, which the guard ignores + + +@contextlib.contextmanager +def _fake_vllm( + present, + broken, + error = _LIBCUDART_ERROR, +): + """Install a synthetic lazy vLLM, restoring VLLM_BROKEN, find_spec, + meta_path, and the vllm* sys.modules entries on exit.""" + from unsloth import import_fixes + + submodules = import_fixes._VLLM_COMPILED_EXTENSIONS + saved_meta_path = list(sys.meta_path) + saved_find_spec = importlib.util.find_spec + saved_broken = import_fixes.VLLM_BROKEN + saved_modules = {n: sys.modules.get(n) for n in ("vllm", *submodules)} + try: + import_fixes.VLLM_BROKEN = False + fake_vllm = types.ModuleType("vllm") + fake_vllm.__path__ = [] + fake_vllm.__spec__ = importlib.machinery.ModuleSpec("vllm", loader = None, is_package = True) + sys.modules["vllm"] = fake_vllm + for name in submodules: + sys.modules.pop(name, None) + sys.meta_path.insert(0, _FakeVllmFinder(present, broken, error)) + yield import_fixes + finally: + import_fixes.VLLM_BROKEN = saved_broken + sys.meta_path[:] = saved_meta_path + importlib.util.find_spec = saved_find_spec + for name, module in saved_modules.items(): + if module is None: + sys.modules.pop(name, None) + else: + sys.modules[name] = module + + +@pytest.mark.parametrize( + "broken_ext", + ["vllm._C", "vllm._C_stable_libtorch"], + ids = ["core_C", "sibling_C_stable_libtorch"], +) +def test_disable_broken_vllm_detects_lazy_loaded_broken_extension(broken_ext): + # A CUDA-major mismatch breaks every ext; whichever one loads first must trip detection. + present = {"vllm._C", "vllm._C_stable_libtorch"} + with _fake_vllm(present = present, broken = {broken_ext}) as import_fixes: + detected = import_fixes.disable_broken_vllm() + + assert detected is True, ( + f"disable_broken_vllm missed an ABI-broken {broken_ext} behind a " + "lazily-importable vllm package — issue #6590 would resurface." + ) + assert import_fixes.VLLM_BROKEN is True + # Once disabled, vLLM must look absent so callers fall back cleanly. + assert importlib.util.find_spec("vllm") is None + + +@pytest.mark.parametrize( + "error", + [ + "libnccl.so.2: cannot open shared object file: No such file or directory", + "libcuda.so.1: cannot open shared object file: No such file or directory", + ], + ids = ["libnccl", "libcuda"], +) +def test_disable_broken_vllm_detects_non_cudart_so_failure(error): + # A CUDA mismatch can surface through a non-libcudart .so (libnccl, libcuda), + # which the old libcudart/libcublas/libnvrtc allow-list let slip through. + with _fake_vllm(present = {"vllm._C"}, broken = {"vllm._C"}, error = error) as import_fixes: + detected = import_fixes.disable_broken_vllm() + + assert detected is True, ( + f"disable_broken_vllm missed a present-but-broken vllm._C raising " + f"{error!r} — vLLM would be left enabled and crash later." + ) + assert import_fixes.VLLM_BROKEN is True + + +@pytest.mark.parametrize( + "present", + [{"vllm._C"}, {"vllm._C", "vllm._C_stable_libtorch", "vllm._moe_C"}], + ids = ["core_only", "all_present"], +) +def test_disable_broken_vllm_keeps_healthy_vllm_enabled(present): + # Healthy install: an absent sibling (ModuleNotFoundError) or an extra present + # ext that loads cleanly must NOT be mistaken for an ABI break. + with _fake_vllm(present = present, broken = set()) as import_fixes: + detected = import_fixes.disable_broken_vllm() + + assert detected is False + assert import_fixes.VLLM_BROKEN is False + assert importlib.util.find_spec("vllm") is not None + + +if __name__ == "__main__": + raise SystemExit(pytest.main([__file__, "-v"])) diff --git a/unsloth/_gpu_init.py b/unsloth/_gpu_init.py index 917717e08e..7f080336aa 100644 --- a/unsloth/_gpu_init.py +++ b/unsloth/_gpu_init.py @@ -36,16 +36,15 @@ from .import_fixes import ( fix_huggingface_hub, ) -# Redirect a read-only Hugging Face cache before anything below can import -# huggingface_hub / transformers / vllm (disable_broken_vllm probes -# `import vllm`, check_fbgemm_gpu_version imports transformers, and -# fix_huggingface_hub imports huggingface_hub itself), all of which can -# freeze Hub's cache constants with the un-redirected paths. unsloth_zoo -# runs the same redirect at import, but that happens after these probes. -# hf_cache.py is stdlib-only, so load it straight from its file without -# triggering the full unsloth_zoo package init this early; the zoo's own -# call later is an idempotent no-op. Older unsloth_zoo without hf_cache.py -# is skipped silently. +# Redirect a read-only Hugging Face cache before anything below imports +# huggingface_hub / transformers / vllm (disable_broken_vllm probes `import vllm` +# and its compiled extensions, check_fbgemm_gpu_version imports transformers, +# fix_huggingface_hub imports huggingface_hub) -- any of which would freeze Hub's +# cache constants with the un-redirected paths. unsloth_zoo runs the same redirect +# at import, but only after these probes. hf_cache.py is stdlib-only, so load it +# straight from its file without triggering the full unsloth_zoo init this early; +# the zoo's later call is an idempotent no-op. Older unsloth_zoo without it is +# skipped silently. try: import importlib.util as _importlib_util from pathlib import Path as _Path diff --git a/unsloth/import_fixes.py b/unsloth/import_fixes.py index bff55b4e7b..d979e5f22f 100644 --- a/unsloth/import_fixes.py +++ b/unsloth/import_fixes.py @@ -2354,11 +2354,9 @@ def _is_broken_vllm_error(error) -> bool: ) ) or ("vllm" in message and "undefined symbol" in message): return True - # Also catch CUDA shared library mismatches during vllm import - # e.g. "libcudart.so.12: cannot open shared object file" - if ( - "libcudart" in message or "libcublas" in message or "libnvrtc" in message - ) and "cannot open shared object file" in message: + # Forced extension load raises the bare loader error (no "vllm._C" + # wrapper); match any .so failure as callers feed only vLLM imports. + if "cannot open shared object file" in message: return True current = getattr(current, "__cause__", None) or getattr(current, "__context__", None) return False @@ -2550,6 +2548,16 @@ def _clear_vllm_modules(): sys.modules.pop(module_name, None) +# vLLM's compiled extensions. A CUDA-major ABI break hits all of them, so +# probing the eagerly-loaded _C and its siblings reliably trips it. +_VLLM_COMPILED_EXTENSIONS = ( + "vllm._C", + "vllm._C_stable_libtorch", + "vllm._moe_C", + "vllm._rocm_C", +) + + def disable_broken_vllm(error = None): """Disable vLLM dynamically when its shared library is ABI-broken.""" global VLLM_BROKEN @@ -2567,6 +2575,15 @@ def disable_broken_vllm(error = None): try: import vllm # noqa: F401 + + # Lazy vLLM lets a bare `import vllm` succeed even when an extension + # is ABI-broken; force-load each to surface the .so failure here. + # A missing one raises ModuleNotFoundError (skipped below). + for _ext in _VLLM_COMPILED_EXTENSIONS: + try: + importlib.import_module(_ext) + except ModuleNotFoundError: + pass return False except Exception as import_error: failure = import_error