Studio: refine GGUF per-GPU selection (gpu_ids) (#7239)

---------

Co-authored-by: oobabooga <112222186+oobabooga@users.noreply.github.com>
This commit is contained in:
Daniel Han 2026-07-23 21:02:29 -07:00 committed by GitHub
commit a7761e1740
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
12 changed files with 771 additions and 216 deletions

View file

@ -2023,6 +2023,10 @@ class LlamaCppBackend:
self._tensor_split: Optional[List[float]] = None
# User-picked physical GPU indices (None = automatic selection).
self._gpu_ids: Optional[List[int]] = None
# RAW requested GPU pin, before the fit narrowed it. self._gpu_ids records the
# EFFECTIVE (fit-narrowed) pin for /status; dedupe compares this raw value so a
# [0, 1] narrowed to [0] and re-sent as [0, 1] still matches (#7239).
self._requested_gpu_ids: Optional[List[int]] = None
# Layer load kept multi-GPU only to honor a downgraded tensor request, so a
# later explicit tensor-off reloads instead of deduping to it (#6659).
self._layer_preserves_tensor_intent: bool = False
@ -2494,6 +2498,46 @@ class LlamaCppBackend:
"""User-picked physical GPU indices, or None for automatic selection."""
return self._gpu_ids
@property
def requested_gpu_ids(self) -> Optional[List[int]]:
"""RAW requested GPU pin (before the fit narrowed it), or None for auto.
gpu_ids echoes the EFFECTIVE pin for /status."""
return self._requested_gpu_ids
def matches_gpu_ids(self, gpu_ids: Optional[List[int]]) -> bool:
"""Whether a requested pin is already satisfied by the active runner.
A regular GGUF load may narrow the requested placement pool to the
smallest fitting subset. Accept both the original request and the
effective status-echoed subset so either can round-trip without a
needless reload. Diffusion drives one device and keeps its existing
lowest-device normalization.
"""
if self._is_diffusion:
requested = [sorted(int(x) for x in gpu_ids)[0]] if gpu_ids else None
return requested == (self._gpu_ids or None)
requested = sorted(int(x) for x in gpu_ids) if gpu_ids else None
raw = self._requested_gpu_ids or None
effective = self._gpu_ids or None
return requested == raw or requested == effective
def _record_matching_gpu_request(self, gpu_ids: Optional[List[int]]) -> None:
"""Adopt the caller's explicit pool after a full already-loaded match.
Matching an effective subset avoids a reload, but the incoming request
is still the user's latest placement intent. Record it so status and a
later reload do not restore GPUs the user just removed.
"""
if self._is_diffusion:
self._requested_gpu_ids = [sorted(int(x) for x in gpu_ids)[0]] if gpu_ids else None
else:
self._requested_gpu_ids = sorted(int(x) for x in gpu_ids) if gpu_ids else None
if self._last_load_kwargs is not None:
self._last_load_kwargs["gpu_ids"] = (
list(self._requested_gpu_ids) if self._requested_gpu_ids else None
)
@property
def n_layers(self) -> Optional[int]:
"""Model layer count (GGUF block_count), or None if unknown."""
@ -4581,6 +4625,14 @@ class LlamaCppBackend:
LlamaCppBackend._gguf_skip_value(f, atype)
return None
@classmethod
def _gguf_path_is_diffusion(cls, gguf_path: str, model_identifier: str) -> bool:
"""Classify a downloaded GGUF without mutating the active backend."""
probe = object.__new__(cls)
probe._model_identifier = model_identifier
probe._read_gguf_metadata(gguf_path)
return probe._is_diffusion
def _read_gguf_metadata(self, gguf_path: str) -> None:
"""Read context_length, architecture params, and chat_template from a GGUF header.
@ -5032,11 +5084,14 @@ class LlamaCppBackend:
# the unload reset) so /status doesn't misreport TP and an identical
# re-Apply doesn't reload against stale tensor-parallel state.
self._tensor_parallel = False
# Record only the single device the runner actually uses (the lowest
# selected GPU, chosen above) -- not the whole pick. The diffusion runner
# is single-device, so echoing a multi-GPU list would misreport placement
# in /status and let a re-Apply dedup against GPUs the runner never used.
# The single-device runner records only the lowest selected GPU (chosen
# above), not the whole pick, and clears any explicit pin from a prior
# chat load; a multi-GPU list would misreport placement and mis-dedup.
self._gpu_ids = [sorted(gpu_ids)[0]] if gpu_ids else None
# The frontend prefers requested_gpu_ids when hydrating the picker.
# Diffusion uses only one device, so echo the collapsed effective pin,
# not unused members of the original request.
self._requested_gpu_ids = list(self._gpu_ids) if self._gpu_ids else None
if hf_variant:
self._hf_variant = hf_variant
elif gguf_path:
@ -6161,6 +6216,8 @@ class LlamaCppBackend:
gpu_layers: int = -1,
n_cpu_moe: int = 0,
tensor_split: Optional[List[float]] = None,
# Explicit GPU placement pool (issue #7164). None/[] = auto-select;
# the fitter may pin the smallest subset of this pool that fits.
gpu_ids: Optional[List[int]] = None,
n_threads: Optional[int] = None,
n_gpu_layers: Optional[int] = None, # caller compat, unused
@ -6258,15 +6315,63 @@ class LlamaCppBackend:
self._cancel_event.clear()
# ── Phase 1: kill old process (under lock, fast) ──────────
with self._lock:
self._kill_process()
# Resolve llama-server now but defer a not-found error: a block-diffusion
# GGUF uses the diffusion runner, and its arch is only known after the header.
binary = self._find_llama_server_binary()
is_vulkan_backend = self._is_vulkan_backend(binary)
# ── Vulkan-ordinal preflight (BEFORE the Phase 1 kill) ────────
# An explicit Vulkan pin the ggml probe never enumerated cannot be honored.
# Validate it ABOVE the kill so an invalid selection leaves the live model
# untouched: CUDA ids are range-checked at the route, but Vulkan ordinals are
# not, so a stale gpu_ids=[99] used to kill the server then 400, leaving
# nothing running (#7239). _get_gpu_memory needs only the binary (safe pre-
# download) and reuses the later fit's issubset logic. Guarded on a found
# Vulkan build + a pin so a deferred not-found stays deferred for diffusion.
if is_vulkan_backend and gpu_ids and binary:
_pf_wanted = {int(x) for x in gpu_ids}
_pf_probed = {g[0] for g in self._get_gpu_memory(binary)}
if not _pf_wanted.issubset(_pf_probed):
raise ValueError(
f"Requested Vulkan GPU ordinal(s) {sorted(_pf_wanted)} not "
f"present. Available Vulkan devices: {sorted(_pf_probed)}."
)
# A remote uncached GGUF may only reveal that it needs the
# single-device diffusion runner after download. On Vulkan, an
# explicit gpu_ids request cannot be mapped from ggml ordinals to
# that runner's CUDA physical index. Download and classify the main
# file before killing the healthy server so this late rejection is
# non-destructive. The Phase 2 call below reuses this cached path.
_preflight_model_path = None
if is_vulkan_backend and gpu_ids and hf_repo:
_resolved_repo = _resolve_repo_id_casing(hf_repo)
if _resolved_repo != hf_repo:
logger.info(
"Using cached repo_id casing '%s' for requested '%s'",
_resolved_repo,
hf_repo,
)
hf_repo = _resolved_repo
with _hf_offline_if_dns_dead():
_preflight_model_path = self._download_gguf(
hf_repo = hf_repo,
hf_variant = hf_variant,
hf_token = hf_token,
)
if self._gguf_path_is_diffusion(_preflight_model_path, model_identifier):
raise ValueError(
"GPU selection (gpu_ids) is not supported for a DiffusionGemma "
"GGUF on a Vulkan llama.cpp build: the diffusion runner selects "
"its device by CUDA physical index, which has no defined mapping "
"to ggml Vulkan device ordinals. Omit gpu_ids to use the default "
"device."
)
# ── Phase 1: kill old process (under lock, fast) ──────────
with self._lock:
self._kill_process()
# ── Phase 2: download (NO lock held, so cancel can proceed) ──
# mtp_draft_path arrives set for local Gemma loads (detected
# sibling); for -hf loads it's None here and resolved just below.
@ -6288,7 +6393,7 @@ class LlamaCppBackend:
)
hf_repo = _resolved_repo
with _hf_offline_if_dns_dead():
model_path = self._download_gguf(
model_path = _preflight_model_path or self._download_gguf(
hf_repo = hf_repo,
hf_variant = hf_variant,
hf_token = hf_token,
@ -6338,6 +6443,18 @@ class LlamaCppBackend:
# Block-diffusion GGUFs (DiffusionGemma) cannot run on llama-server;
# serve them with the diffusion runner (same OpenAI-compat interface).
if self._is_diffusion:
# The diffusion runner pins its child by CUDA visibility mask, so a
# ggml Vulkan ordinal cannot be honored (wrong GPU / CPU fallback).
# Route and remote-download preflights reject before teardown; keep
# this as a final defense if classification ever disagrees.
if is_vulkan_backend and gpu_ids:
raise ValueError(
"GPU selection (gpu_ids) is not supported for a DiffusionGemma "
"GGUF on a Vulkan llama.cpp build: the diffusion runner selects "
"its device by CUDA physical index, which has no defined mapping "
"to ggml Vulkan device ordinals. Omit gpu_ids to use the default "
"device."
)
# Not a tensor/layer GGUF: clear any preserved-fallback flag from a
# prior load (this path skips the command builder that clears it).
self._layer_preserves_tensor_intent = False
@ -6558,6 +6675,12 @@ class LlamaCppBackend:
# Layer-fallback min GPUs; raised below on a tensor downgrade. Bound
# before the try so the --fit-on except path still has it (no UnboundLocal).
_layer_min_gpus = 1
# An explicit Vulkan ordinal absent from the ggml probe cannot be
# honored; flag it in the fit and reject after the try (raising inside
# would be swallowed into the --fit-on fallback). Bound before the try.
_vulkan_explicit_unmatched = False
_vulkan_requested_ids: list[int] = []
_vulkan_available_ordinals: list[int] = []
try:
gguf_size = self._get_gguf_size_bytes(model_path)
# Include GPU-loaded mmproj in the fit budget (#5825).
@ -6570,6 +6693,28 @@ class LlamaCppBackend:
# Pass binary so a Vulkan build probes ggml's Vulkan ordinals.
_gpu_mem = self._get_gpu_memory(binary)
gpus = [(idx, free) for idx, free, _t in _gpu_mem]
# Restrict the fit (and thus the layer plan + pin env) to the
# selected GPUs; fail-open if none match so a stale UI choice
# can't strand the load on CPU (issue #7164).
if gpu_ids:
# A Vulkan build indexes by ggml ordinal. An explicit ordinal
# absent from the probe can't be pinned, so reject after the try
# rather than fail-open onto a device the user didn't pick.
_wanted_ids = {int(x) for x in gpu_ids}
# Reject if ANY requested ordinal is absent, not only when none
# match: [0, 99] against {0, 1} silently drops 99. Comparing the
# full requested set (before filter narrows) still lets the fitter
# pick a valid subset later -- that is narrowing, not absence.
_probed_ordinals = {g[0] for g in gpus}
if is_vulkan_backend and not _wanted_ids.issubset(_probed_ordinals):
_vulkan_explicit_unmatched = True
_vulkan_requested_ids = sorted(_wanted_ids)
_vulkan_available_ordinals = sorted(_probed_ordinals)
# Restrict the probed pool to the selection; fail-open (keep the
# full pool) if none match so a stale UI choice can't strand the
# load on CPU (issue #7164).
_sel_gpus = [g for g in gpus if g[0] in _wanted_ids]
gpus = _sel_gpus if _sel_gpus else gpus
total_by_idx = {idx: total for idx, _f, total in _gpu_mem}
# GPU picker: restrict every mode to the chosen devices, so
# auto selection only considers them and manual mask to
@ -7396,6 +7541,17 @@ class LlamaCppBackend:
tp_tensor_split = None
effective_ctx = requested_ctx # fall back to original
# An unenumerated explicit Vulkan ordinal can't be pinned; fail loudly
# instead of fitting onto an unselected device. Clear the raw selection
# the early state-publish recorded so it never leaks into gpu_ids (#7239).
if _vulkan_explicit_unmatched:
self._gpu_ids = None
self._requested_gpu_ids = None
raise ValueError(
f"Requested Vulkan GPU ordinal(s) {_vulkan_requested_ids} not "
f"present. Available Vulkan devices: {_vulkan_available_ordinals}."
)
# GPU picker: when no narrower subset was chosen (manual, or
# a failed/file-size selection), pin the whole picked set so the
# model can't spill onto an unpicked GPU.
@ -7759,11 +7915,45 @@ class LlamaCppBackend:
", ".join(unsupported_cache_flags),
)
# Vulkan pins via --device (a cmd arg, unlike the env-based
# CUDA/ROCm pin below), emitted BEFORE user extras so llama.cpp's
# last-wins parsing lets a user --device override Unsloth's pick.
if is_vulkan_backend and gpu_indices is not None:
cmd += LlamaCppBackend._vulkan_pin_args(gpu_indices)
# Vulkan pins via --device (a cmd arg), before user extras so a user
# --device wins. Fall back to raw ids when the fit did not narrow.
_vulkan_pin_ids = gpu_indices if gpu_indices is not None else (gpu_ids or None)
# Record the pin actually applied (fit-narrowed gpu_indices, else the raw
# request) for the keep-warm loop, dedupe, and /status, so an explicit
# [0, 1] narrowed to [0] records [0] and /status never echoes an ordinal
# the child never saw. Auto selection (no gpu_ids) stays None (#7239).
if is_vulkan_backend:
# Only record an EXPLICIT Vulkan pin: an auto pick still narrows +
# pins below, but recording it would misreport an explicit pin and
# make dedupe miss the loaded server; mirrors the CUDA/ROCm branch.
self._gpu_ids = (
sorted(int(x) for x in _vulkan_pin_ids)
if (gpu_ids and _vulkan_pin_ids)
else None
)
elif gpu_ids:
# Physical pin: the fit-selected subset when the fit ran, else the raw
# user selection so an explicit choice is honoured even when the fit
# could not size the model.
_effective_pin_ids = (
[int(x) for x in gpu_indices]
if gpu_indices is not None
else [int(x) for x in gpu_ids]
)
self._gpu_ids = (
sorted(int(x) for x in _effective_pin_ids) if _effective_pin_ids else None
)
else:
self._gpu_ids = None
# Also record the RAW requested pin (before the fit narrowed it). Load
# dedupe compares this so a [0, 1] narrowed to [0] and re-sent as [0, 1]
# still matches, while /status keeps echoing the effective pin (#7239).
self._requested_gpu_ids = sorted(int(x) for x in gpu_ids) if gpu_ids else None
if is_vulkan_backend and _vulkan_pin_ids is not None:
cmd += LlamaCppBackend._vulkan_pin_args(_vulkan_pin_ids)
# User pass-through args go last so llama.cpp's last-wins parsing
# lets the user override Unsloth's auto-set flags. Already
@ -7832,10 +8022,10 @@ class LlamaCppBackend:
f"Data-center GPU detected: applied DC llama.cpp env tuning (multi_gpu={multi_gpu})"
)
# Pin to selected GPU(s). On ROCm, narrowing only
# CUDA_VISIBLE_DEVICES leaves an AMD child seeing the full set, so
# set HIP_VISIBLE_DEVICES too. Vulkan is pinned via --device
# (above), not here.
# Pin to selected GPU(s) (issue #7164; resolved above into gpu_indices).
# On ROCm, narrowing only CUDA_VISIBLE_DEVICES leaves the AMD child
# seeing the full set, so set HIP_VISIBLE_DEVICES too. Vulkan is pinned
# via --device (above), not here.
# A deliberate zero-offload load with no GPU companions runs
# entirely on CPU, yet a visible CUDA device still costs the child
# ~0.5 GB (context + compute scratch) that the CPU-only
@ -8756,16 +8946,10 @@ class LlamaCppBackend:
)
):
return False
# A changed GPU pick must reload (compare order-insensitively; None/[]
# both mean automatic). The diffusion runner collapses a multi-GPU pick
# to its single lowest device, so self._gpu_ids holds just that device;
# normalize the request the same way, or a multi-GPU pick that resolves
# to the same device needlessly reloads.
if self._is_diffusion:
requested_gpu_pick = [sorted(gpu_ids)[0]] if gpu_ids else None
else:
requested_gpu_pick = sorted(gpu_ids) if gpu_ids else None
if (self._gpu_ids or None) != requested_gpu_pick:
# A changed GPU pick must reload. Regular GGUF accepts either the raw
# requested placement pool or the effective status-echoed subset;
# diffusion compares its normalized single-device pick.
if not self.matches_gpu_ids(gpu_ids):
return False
# Compare on the canonical requested mode. With --spec-type in
@ -8823,6 +9007,7 @@ class LlamaCppBackend:
current = list(self._extra_args) if self._extra_args is not None else []
if list(extra_args) != current:
return False
self._record_matching_gpu_request(gpu_ids)
return True
def _classify_gpu_offload(
@ -8954,12 +9139,15 @@ class LlamaCppBackend:
self._supports_preserve_thinking = False
self._supports_tools = False
self._cache_type_kv = None
# GPU-pin state describes the active runner only; clear it so an explicit
# pin never leaks into the next (or diffusion) runner.
self._gpu_ids = None
self._requested_gpu_ids = None
self._tensor_parallel = False
self._gpu_memory_mode = "auto"
self._gpu_layers = -1
self._n_cpu_moe = 0
self._tensor_split = None
self._gpu_ids = None
self._layer_preserves_tensor_intent = False
self._speculative_type = None
self._requested_spec_mode = None

View file

@ -74,7 +74,7 @@ class LoadRequest(BaseModel):
)
gpu_ids: Optional[List[int]] = Field(
None,
description = "Physical GPU indices to use, for example [0, 1]. Omit or pass [] to use automatic selection. Explicit gpu_ids are unsupported when the parent CUDA_VISIBLE_DEVICES uses UUID/MIG entries. For GGUF models the picked devices are pinned via CUDA/HIP_VISIBLE_DEVICES.",
description = "GPU placement pool, for example [0, 1]. Omit or pass [] to use automatic selection. CUDA/ROCm values are physical GPU indices and are unsupported when the parent CUDA_VISIBLE_DEVICES uses UUID/MIG entries; Vulkan values are ggml device ordinals. For GGUF models the fitter may pin the smallest subset of this pool that fits.",
)
speculative_type: Optional[str] = Field(
None,
@ -485,7 +485,14 @@ class LoadResponse(BaseModel):
)
gpu_ids: Optional[List[int]] = Field(
None,
description = "Physical GPU indices the model is pinned to, or None for automatic selection.",
description = "Effective GPU indices the model is using after fit-time narrowing, or None for automatic selection.",
)
requested_gpu_ids: Optional[List[int]] = Field(
None,
description = (
"GPU placement pool requested by the user before fit-time narrowing, "
"or None for automatic selection."
),
)
@ -649,7 +656,14 @@ class InferenceStatusResponse(BaseModel):
)
gpu_ids: Optional[List[int]] = Field(
None,
description = "Physical GPU indices the model is pinned to, or None for automatic selection.",
description = "Effective GPU indices the model is using after fit-time narrowing, or None for automatic selection.",
)
requested_gpu_ids: Optional[List[int]] = Field(
None,
description = (
"GPU placement pool requested by the user before fit-time narrowing, "
"or None for automatic selection."
),
)
llama_cpp_supports_mtp: bool = Field(
True,

View file

@ -3241,15 +3241,10 @@ def _request_matches_loaded_settings(
)
):
return False
# A changed GPU pick must reload. The diffusion runner collapses a multi-GPU
# request to its single lowest device (it drives one device only), so the
# backend records just that device; compare the request the same way, or a
# multi-GPU pick that resolves to the same device needlessly reloads.
if llama_backend.is_diffusion:
_req_gpu_ids = [sorted(request.gpu_ids)[0]] if request.gpu_ids else None
else:
_req_gpu_ids = sorted(request.gpu_ids) if request.gpu_ids else None
if _req_gpu_ids != llama_backend.gpu_ids:
# A regular GGUF may narrow the requested placement pool. Accept either the
# original request or the effective status-echoed subset; diffusion keeps
# its single-device normalization.
if not llama_backend.matches_gpu_ids(request.gpu_ids):
return False
# Preserved tensor->layer fallback (both report tensor=off, so the check above
# matches): if the user now explicitly drops tensor intent, reload so placement
@ -3897,15 +3892,19 @@ def _classify_diffusion_gguf(config: ModelConfig) -> Optional[bool]:
"""Classify a GGUF as diffusion, normal, or unknown before it is loaded.
``None`` is important here: a remote GGUF whose header is not cached can
still be routed to the single-GPU diffusion runner after download. Treating
that case as normal would let Manual mode skip the training guard even
though the runner ignores Manual's llama-server placement controls.
still be routed to the single-GPU diffusion runner after download. Default
placement keeps that unknown case guarded until the header is available.
"""
identity = " ".join(
str(getattr(config, attr, "") or "") for attr in ("identifier", "gguf_hf_repo", "gguf_file")
).lower()
if "diffusion" in identity:
return True
# Name-only hint, used ONLY as a pre-download fallback, scoped to the
# DiffusionGemma runner family: a bare "diffusion" substring is common in
# ordinary text-model names/paths (e.g. "stable-diffusion-prompt"), and treating
# those as diffusion falsely rejects a valid Vulkan+gpu_ids GGUF (#7239). Normalize
# non-alphanumerics so "DiffusionGemma"/"diffusion-gemma" collapse to one token.
# The local header below stays authoritative.
name_says_diffusion = "diffusiongemma" in _re.sub(r"[^a-z0-9]+", "", identity)
try:
main = getattr(config, "gguf_file", None)
@ -3915,23 +3914,86 @@ def _classify_diffusion_gguf(config: ModelConfig) -> Optional[bool]:
if repo and variant:
from hub.utils.gguf import resolve_local_gguf_path
main = resolve_local_gguf_path(repo, variant)
if not main or not Path(main).is_file():
return None
probe = LlamaCppBackend()
probe._read_gguf_metadata(str(main))
if probe.is_diffusion:
return True
# A successfully decoded architecture proves that this is a normal
# llama-server GGUF. No architecture means the lightweight probe could
# not establish the routing decision, so preserve the unknown state.
if getattr(probe, "_architecture", None):
return False
return None
if main and Path(main).is_file():
# The local GGUF header is authoritative (same probe the loader uses), so
# it can't be fooled by a "diffusion"-flavored name/path.
probe = LlamaCppBackend()
probe._read_gguf_metadata(str(main))
if probe.is_diffusion:
return True
# A decoded architecture proves a normal llama-server GGUF; no architecture
# means the probe was inconclusive, so fall through to the name hint below.
if getattr(probe, "_architecture", None):
return False
except Exception as e:
logger.debug("Could not identify diffusion GGUF for training guard: %s", e)
# Header unavailable (remote uncached) or inconclusive: True only for the
# DiffusionGemma name family; otherwise None keeps an unknown remote GGUF guarded
# as potentially diffusion until its header proves otherwise.
return True if name_says_diffusion else None
async def _resolve_gguf_gpu_ids_for_request(
config: ModelConfig, gpu_ids: Optional[List[int]]
) -> Optional[List[int]]:
"""Resolve and fully validate an explicit GGUF GPU placement pool.
CUDA and ROCm use physical IDs. Vulkan uses ggml ordinals, so its device
existence check comes from the same ggml probe used by the loader. Both
/load and /validate call this before their training guard or any teardown.
"""
if not gpu_ids:
return None
from utils.hardware import DeviceType, get_device
from utils.hardware.hardware import resolve_requested_gpu_ids
is_vulkan = LlamaCppBackend._is_vulkan_backend()
if get_device() == DeviceType.XPU and not is_vulkan:
raise HTTPException(
status_code = 400,
detail = (
"GPU selection (gpu_ids) is not supported on Intel XPU. "
"Omit gpu_ids to use all devices."
),
)
if is_vulkan and _classify_diffusion_gguf(config) is True:
raise HTTPException(
status_code = 400,
detail = (
"GPU selection (gpu_ids) is not supported for a DiffusionGemma "
"GGUF on a Vulkan llama.cpp build: the diffusion runner selects "
"its device by CUDA physical index, which has no defined mapping "
"to ggml Vulkan device ordinals. Omit gpu_ids to use the default "
"device."
),
)
try:
resolved = resolve_requested_gpu_ids(gpu_ids, is_vulkan = is_vulkan)
except ValueError as exc:
raise HTTPException(status_code = 400, detail = str(exc)) from exc
if is_vulkan and resolved:
binary = LlamaCppBackend._find_llama_server_binary()
if binary:
probed = {
gpu[0] for gpu in await asyncio.to_thread(LlamaCppBackend._get_gpu_memory, binary)
}
wanted = {int(gpu_id) for gpu_id in resolved}
if not wanted.issubset(probed):
raise HTTPException(
status_code = 400,
detail = (
f"Requested Vulkan GPU ordinal(s) {sorted(wanted)} not "
f"present. Available Vulkan devices: {sorted(probed)}."
),
)
return resolved
def _guard_chat_load_against_training(
config: ModelConfig,
@ -3971,8 +4033,18 @@ def _guard_chat_load_against_training(
if is_gguf and gpu_memory_mode == "manual" and diffusion_kind is False:
return
# Vulkan GGUF pins are ggml ordinals, not CUDA physical IDs. Detect this
# before deriving a possible diffusion fallback device so an unknown remote
# GGUF never sends its ordinal through the CUDA single-device path.
is_vulkan = False
if is_gguf:
try:
is_vulkan = LlamaCppBackend._is_vulkan_backend()
except Exception as e:
logger.warning("Could not detect Vulkan backend for chat-load guard: %s", e)
diffusion_gpu = None
if is_gguf and diffusion_kind is not False:
if is_gguf and diffusion_kind is not False and not (is_vulkan and requested_gpu_ids):
# Use the same token selection as the runner: an explicit pick wins,
# followed by DG_GPU, the first parent-visible token, then GPU 0.
diffusion_gpu = LlamaCppBackend._diffusion_gpu_arg(
@ -3999,6 +4071,7 @@ def _guard_chat_load_against_training(
max_seq_length = max_seq_length,
requested_gpu_ids = requested_gpu_ids,
is_gguf = is_gguf,
is_vulkan = is_vulkan,
required_override_gb = required_override_gb,
single_device_gpu = diffusion_gpu,
)
@ -4305,6 +4378,7 @@ async def _load_model_impl(
# Skip if a prior audio probe failed -- let load_model retry.
and getattr(llama_backend, "_audio_probed", True)
):
llama_backend._record_matching_gpu_request(request.gpu_ids)
logger.info(
"Model already loaded (GGUF): "
f"{model_log_label} variant={request.gguf_variant or llama_backend.hf_variant}, skipping reload"
@ -4351,6 +4425,7 @@ async def _load_model_impl(
n_layers = llama_backend.n_layers,
n_moe_layers = llama_backend.n_moe_layers,
gpu_ids = llama_backend.gpu_ids,
requested_gpu_ids = llama_backend.requested_gpu_ids,
)
else:
if (
@ -4417,41 +4492,12 @@ async def _load_model_impl(
# Normalize gpu_ids: empty list means auto-selection, same as None
effective_gpu_ids = request.gpu_ids if request.gpu_ids else None
# GGUF supports gpu_ids: validate the pick up front (before the training
# guard) so a bad pick is a clean 400, not masked by a VRAM 409. Rejects
# negative / out-of-range / duplicate ids and UUID/MIG parents. XPU hosts
# are rejected outright: the picker's indices are torch-xpu ordinals neither
# applicator speaks (CUDA/HIP masks don't apply, the Vulkan --device pin
# uses ggml's own Vulkan ordinals), so a pick could land on the wrong device.
if config.is_gguf and effective_gpu_ids is not None:
from utils.hardware import DeviceType, get_device
from utils.hardware.hardware import resolve_requested_gpu_ids
if get_device() == DeviceType.XPU:
raise HTTPException(
status_code = 400,
detail = (
"GPU selection (gpu_ids) is not supported on Intel XPU. "
"Omit gpu_ids to use all devices."
),
)
# Same reasoning for a Vulkan-only build: --device pins ggml's own
# Vulkan ordinals, so a physical pick can land on the wrong card on
# masked or non-contiguous hosts.
if LlamaCppBackend._is_vulkan_backend():
raise HTTPException(
status_code = 400,
detail = (
"GPU selection (gpu_ids) is not supported with a Vulkan "
"llama.cpp build: physical GPU ids have no defined "
"mapping to Vulkan device ordinals. Omit gpu_ids to use "
"all devices."
),
)
try:
resolve_requested_gpu_ids(effective_gpu_ids)
except ValueError as exc:
raise HTTPException(status_code = 400, detail = str(exc)) from exc
# Validate the full GGUF placement pool before the training guard so an
# invalid physical ID or Vulkan ordinal is a clean 400, not a masked VRAM
# 409. The same helper is used by /validate.
gguf_gpu_ids: Optional[List[int]] = None
if config.is_gguf:
gguf_gpu_ids = await _resolve_gguf_gpu_ids_for_request(config, effective_gpu_ids)
if not config.is_gguf and _mlx_distributed_launch_detected():
raise HTTPException(
status_code = 400,
@ -4575,8 +4621,9 @@ async def _load_model_impl(
gpu_layers = request.gpu_layers,
n_cpu_moe = request.n_cpu_moe,
tensor_split = request.tensor_split,
gpu_ids = effective_gpu_ids,
n_parallel = _n_parallel,
# Issue #7164: explicit GPU pin resolved to physical ids above.
gpu_ids = gguf_gpu_ids,
)
if config.gguf_hf_repo:
# HF mode: download via huggingface_hub then start llama-server
@ -4750,6 +4797,7 @@ async def _load_model_impl(
n_layers = llama_backend.n_layers,
n_moe_layers = llama_backend.n_moe_layers,
gpu_ids = llama_backend.gpu_ids,
requested_gpu_ids = llama_backend.requested_gpu_ids,
)
# ── Standard path: load via Unsloth/transformers ──────────
@ -5043,36 +5091,8 @@ async def validate_model(
# Apply the same training coexistence policy as /load before the frontend
# unloads the current model.
effective_gpu_ids = request.gpu_ids if request.gpu_ids else None
# Mirror /load: GGUF supports gpu_ids, so validate the pick (a bad one is
# a clean 400) before the guard sizes the model against training VRAM.
# XPU-host picks are rejected like /load (no defined mapping from the
# picker's torch-xpu ordinals to the launcher's device spaces).
if config.is_gguf and effective_gpu_ids is not None:
from utils.hardware import DeviceType, get_device
from utils.hardware.hardware import resolve_requested_gpu_ids
if get_device() == DeviceType.XPU:
raise HTTPException(
status_code = 400,
detail = (
"GPU selection (gpu_ids) is not supported on Intel XPU. "
"Omit gpu_ids to use all devices."
),
)
if LlamaCppBackend._is_vulkan_backend():
raise HTTPException(
status_code = 400,
detail = (
"GPU selection (gpu_ids) is not supported with a Vulkan "
"llama.cpp build: physical GPU ids have no defined "
"mapping to Vulkan device ordinals. Omit gpu_ids to use "
"all devices."
),
)
try:
resolve_requested_gpu_ids(effective_gpu_ids)
except ValueError as exc:
raise HTTPException(status_code = 400, detail = str(exc)) from exc
if config.is_gguf:
await _resolve_gguf_gpu_ids_for_request(config, effective_gpu_ids)
effective_load_in_4bit = _effective_load_in_4bit(config, request.load_in_4bit)
# Both checks cover the [adapter, base] set (matching the scan route and workers):
@ -5897,6 +5917,7 @@ async def get_status(current_subject: str = Depends(get_current_subject)):
n_layers = llama_backend.n_layers,
n_moe_layers = llama_backend.n_moe_layers,
gpu_ids = llama_backend.gpu_ids,
requested_gpu_ids = llama_backend.requested_gpu_ids,
llama_cpp_supports_mtp = _supports_mtp,
spec_fallback_reason = llama_backend.spec_fallback_reason,
llama_cpp_prebuilt_stale = _stale,

View file

@ -225,6 +225,7 @@ def can_load_chat_during_training(
max_seq_length: int,
requested_gpu_ids: Optional[List[int]],
is_gguf: bool = False,
is_vulkan: bool = False,
required_override_gb: Optional[float] = None,
single_device_gpu: Optional[str] = None,
) -> Tuple[bool, Dict[str, Any]]:
@ -233,11 +234,15 @@ def can_load_chat_during_training(
chat model against the free VRAM that remains). Sizes/places it the same way
the loader will: HF auto reuses auto_select_gpu_ids; HF explicit requires an
even-share per-GPU floor for device_map="balanced"; GGUF sizes from
required_override_gb over the visible pool. ``single_device_gpu`` is the
exact physical device token selected by a single-device runner.
`load_in_4bit` must be effective (LoRA can flip 4-bit -> 16-bit). Non-CUDA
allows the load; default-deny on any CUDA case it can't size, so a load never
OOMs training."""
required_override_gb over the visible pool. A Vulkan GGUF selection picks by ggml
Vulkan ordinal (separate index space from CUDA ids), so its requested_gpu_ids is
NOT resolved against the CUDA set (which would raise -> invalid_gpu_ids -> bypass
the OOM check); conservatively size an N-device request against the least-free
N visible GPUs instead.
``single_device_gpu`` is the exact physical device token selected by a
single-device runner. `load_in_4bit` must be effective (LoRA can flip 4-bit
-> 16-bit). Non-CUDA allows the load; default-deny on any CUDA case it can't
size, so a load never OOMs training."""
try:
from utils.hardware import (
DeviceType,
@ -258,6 +263,11 @@ def can_load_chat_during_training(
max_seq_length = max_seq_length or 2048,
)
# A Vulkan GGUF selection uses ggml Vulkan ordinals, not CUDA physical ids;
# size it against the full visible pool (GGUF self-placement) rather than
# resolving ordinals against the CUDA parent-visible set.
vulkan_gguf = is_gguf and is_vulkan
# HF auto: reuse the loader's selector; fits iff its pick clears the margin.
if not requested_gpu_ids and not is_gguf:
_selected, meta = auto_select_gpu_ids(model_name, **est_kwargs)
@ -283,7 +293,9 @@ def can_load_chat_during_training(
}
# Explicit GPUs, or GGUF: size directly and check live free VRAM.
if single_device_gpu is not None:
if requested_gpu_ids and vulkan_gguf:
mode = "gguf_vulkan"
elif single_device_gpu is not None:
mode = "single_device"
elif is_gguf:
mode = "gguf"
@ -296,7 +308,17 @@ def can_load_chat_during_training(
return False, {"mode": mode, "reason": "estimate_unavailable"}
free_by_index = _free_vram_by_index(get_visible_gpu_utilization().get("devices", []))
if single_device_gpu is not None:
if requested_gpu_ids and vulkan_gguf:
# Vulkan ordinals cannot be mapped to CUDA physical indices. Budget
# the least-free N visible cards for an N-device request. If that
# conservative subset fits, any physical mapping of the ordinals
# fits, without collapsing a multi-GPU request to one card.
visible_free = list(free_by_index.values())
if not visible_free:
return False, {"mode": "gguf_vulkan", "reason": "no_visible_gpus"}
n_pins = min(len(requested_gpu_ids), len(visible_free))
free_vals = sorted(visible_free)[:n_pins]
elif single_device_gpu is not None:
token = str(single_device_gpu).strip()
if not token:
# Empty token = a CPU-only single-device runner (e.g. a CPU
@ -324,7 +346,8 @@ def can_load_chat_during_training(
return True, {"mode": mode, "reason": "invalid_gpu_ids"}
free_vals = [free_by_index.get(i, 0.0) for i in resolved]
else:
# GGUF: llama.cpp picks the GPU(s); any visible GPU is a candidate.
# GGUF self-placement / auto Vulkan (no requested ids): llama.cpp picks
# the GPU(s), so any visible GPU is a candidate -> size the whole pool.
free_vals = list(free_by_index.values())
if not free_vals:

View file

@ -170,6 +170,7 @@ class TestCanLoadGGUF(_GpuCacheResetMixin, unittest.TestCase):
estimate = None,
single_device_gpu = None,
gpu_ids = None,
is_vulkan = False,
):
with (
patch("utils.hardware.get_device", return_value = DeviceType.CUDA),
@ -185,6 +186,7 @@ class TestCanLoadGGUF(_GpuCacheResetMixin, unittest.TestCase):
max_seq_length = 0,
requested_gpu_ids = gpu_ids,
is_gguf = True,
is_vulkan = is_vulkan,
required_override_gb = required_override,
single_device_gpu = single_device_gpu,
)
@ -234,6 +236,35 @@ class TestCanLoadGGUF(_GpuCacheResetMixin, unittest.TestCase):
self.assertFalse(blocked)
self.assertEqual(blocked_info["usable_gb"], 10.0)
def test_vulkan_pin_takes_precedence_over_unknown_diffusion_fallback(self):
# An uncached GGUF can carry a speculative single-device fallback while
# its explicit pin is actually a ggml Vulkan ordinal. Never interpret
# that ordinal as the same-numbered CUDA physical device.
ok, info, _ = self._run(
devices = _devices((0, 80, 0), (1, 80, 78)),
required_override = 20.0,
single_device_gpu = "0",
gpu_ids = [0],
is_vulkan = True,
)
self.assertFalse(ok)
self.assertEqual(info["mode"], "gguf_vulkan")
self.assertEqual(info["usable_gb"], 2.0)
def test_vulkan_multi_gpu_guard_counts_requested_devices(self):
# The ordinal mapping is unknown, so use the least-free two visible
# cards for a two-device request. Their aggregate capacity is still
# available instead of collapsing the request to one card.
ok, info, _ = self._run(
devices = _devices((0, 80, 70), (1, 80, 70), (2, 80, 0)),
required_override = 10.0,
gpu_ids = [0, 1],
is_vulkan = True,
)
self.assertTrue(ok)
self.assertEqual(info["mode"], "gguf_vulkan")
self.assertEqual(info["usable_gb"], 18.5)
def test_single_device_unresolved_token_sizes_against_worst_device(self):
# A non-numeric device token (a CUDA UUID / MIG handle) can't map to a
# free-VRAM index. The runner still drives ONE device, so size against the
@ -478,58 +509,19 @@ class TestChatLoadGuardRoute(unittest.TestCase):
def test_manual_known_normal_gguf_bypasses_training_estimate(self):
captured = []
config = SimpleNamespace(is_gguf = True)
with patch.object(self.route, "_classify_diffusion_gguf", return_value = False):
with patch.object(self.route, "_classify_diffusion_gguf", return_value = False) as classify:
self._guard(
config = config,
captured = captured,
training_active = True,
decision = (False, {"reason": "must not run"}),
gpu_memory_mode = "manual",
requested_gpu_ids = [1, 3],
)
classify.assert_called_once_with(config)
self.assertEqual(captured, [])
def test_manual_unknown_gguf_keeps_single_device_training_guard(self):
captured = []
config = SimpleNamespace(is_gguf = True)
with (
patch.object(self.route, "_classify_diffusion_gguf", return_value = None),
patch.object(self.route, "_estimate_gguf_required_gb", return_value = 12.5),
patch.object(
self.route.LlamaCppBackend,
"_diffusion_gpu_arg",
return_value = "2",
),
):
self._guard(
config = config,
captured = captured,
training_active = True,
decision = (True, {"mode": "single_device"}),
gpu_memory_mode = "manual",
)
self.assertEqual(len(captured), 1)
self.assertEqual(captured[0]["single_device_gpu"], "2")
def test_manual_diffusion_uses_single_device_guard(self):
captured = []
config = SimpleNamespace(is_gguf = True)
with (
patch.object(self.route, "_classify_diffusion_gguf", return_value = True),
patch.object(self.route, "_estimate_gguf_required_gb", return_value = 12.5),
):
self._guard(
config = config,
captured = captured,
training_active = True,
decision = (True, {"mode": "gguf"}),
gpu_memory_mode = "manual",
requested_gpu_ids = [3, 1],
)
self.assertEqual(len(captured), 1)
self.assertEqual(captured[0]["single_device_gpu"], "1")
self.assertEqual(captured[0]["requested_gpu_ids"], [3, 1])
def test_unpinned_diffusion_uses_runner_default_gpu(self):
def test_manual_diffusion_keeps_single_device_training_guard(self):
captured = []
config = SimpleNamespace(is_gguf = True)
with (
@ -540,11 +532,6 @@ class TestChatLoadGuardRoute(unittest.TestCase):
"_effective_gpu_count",
return_value = 2,
),
patch.object(
self.route.LlamaCppBackend,
"_diffusion_gpu_arg",
return_value = "3",
) as gpu_arg,
):
self._guard(
config = config,
@ -552,9 +539,11 @@ class TestChatLoadGuardRoute(unittest.TestCase):
training_active = True,
decision = (True, {"mode": "single_device"}),
gpu_memory_mode = "manual",
requested_gpu_ids = [3, 1],
)
gpu_arg.assert_called_once_with(None, cpu_only = False)
self.assertEqual(captured[0]["single_device_gpu"], "3")
self.assertEqual(len(captured), 1)
self.assertEqual(captured[0]["single_device_gpu"], "1")
self.assertEqual(captured[0]["requested_gpu_ids"], [3, 1])
def test_refuses_with_headroom_number(self):
info = {"required_gb": 30.0, "usable_gb": 6.0, "needed_gb": 39.0, "mode": "auto"}

View file

@ -591,10 +591,23 @@ def test_load_request_accepts_gpu_ids():
@pytest.mark.parametrize("model_cls", [LoadResponse, InferenceStatusResponse])
def test_response_models_emit_gpu_ids(model_cls):
if model_cls is LoadResponse:
obj = model_cls(status = "loaded", model = "m", display_name = "m", inference = {}, gpu_ids = [1])
obj = model_cls(
status = "loaded",
model = "m",
display_name = "m",
inference = {},
gpu_ids = [1],
requested_gpu_ids = [1, 2],
)
else:
obj = model_cls(gpu_ids = [1])
obj = model_cls(gpu_ids = [1], requested_gpu_ids = [1, 2])
assert obj.model_dump()["gpu_ids"] == [1]
assert obj.model_dump()["requested_gpu_ids"] == [1, 2]
def test_gguf_load_and_status_responses_include_requested_gpu_pool():
route_src = (Path(_BACKEND_DIR) / "routes" / "inference.py").read_text(encoding = "utf-8")
assert route_src.count("requested_gpu_ids = llama_backend.requested_gpu_ids") == 3
def test_gpu_ids_property_default_and_reset():
@ -625,6 +638,10 @@ def _target_state_gpu_ids(backend, gpu_ids):
def test_gpu_ids_reload_detection_is_order_insensitive():
backend = _loaded_backend("auto")
backend._gpu_ids = [0, 1]
# A real non-narrowed load records the raw request too; the non-diffusion
# dedupe now compares that raw pin (#7239). Set it to match the effective pin
# (no narrowing) so this exercises the order-insensitive comparison.
backend._requested_gpu_ids = [0, 1]
# Same set, different order -> no reload.
assert _target_state_gpu_ids(backend, [1, 0]) is True
# Different set -> reload.
@ -633,6 +650,26 @@ def test_gpu_ids_reload_detection_is_order_insensitive():
assert _target_state_gpu_ids(backend, None) is False
def test_gpu_ids_reload_detection_accepts_raw_and_effective_pin():
backend = _loaded_backend("auto")
backend._requested_gpu_ids = [0, 1]
backend._gpu_ids = [0]
backend._last_load_kwargs = {"gpu_ids": [0, 1], "model_identifier": "owner/repo"}
# The original request still matches after the fitter narrows it.
assert _target_state_gpu_ids(backend, [1, 0]) is True
assert backend.requested_gpu_ids == [0, 1]
# The status response echoes the effective pin, which must also round-trip.
# Treat the incoming subset as the latest intent so status and a future
# reload do not restore GPU 1 after the user removed it.
assert _target_state_gpu_ids(backend, [0]) is True
assert backend.requested_gpu_ids == [0]
assert backend._last_load_kwargs == {"gpu_ids": [0], "model_identifier": "owner/repo"}
# A genuinely different placement pool still reloads.
assert _target_state_gpu_ids(backend, [1]) is False
assert _target_state_gpu_ids(backend, None) is False
def test_gpu_ids_reload_detection_collapses_diffusion_to_single_device():
# The diffusion runner drives only its single lowest device, so the backend
# records [lowest]. A later multi-GPU request that still resolves to that
@ -642,6 +679,7 @@ def test_gpu_ids_reload_detection_collapses_diffusion_to_single_device():
backend._is_diffusion = True
backend._gpu_ids = [1] # loaded on the lowest of an earlier [3, 1] pick
assert _target_state_gpu_ids(backend, [3, 1]) is True
assert backend.requested_gpu_ids == [1]
assert _target_state_gpu_ids(backend, [1]) is True
# Lowest device changes (2, not 1) -> reload.
assert _target_state_gpu_ids(backend, [3, 2]) is False
@ -649,6 +687,56 @@ def test_gpu_ids_reload_detection_collapses_diffusion_to_single_device():
assert _target_state_gpu_ids(backend, None) is False
def test_remote_vulkan_diffusion_preflight_runs_before_teardown(monkeypatch):
def _mark_diffusion(probe, path):
assert path == "/cache/model.gguf"
probe._is_diffusion = True
monkeypatch.setattr(LlamaCppBackend, "_read_gguf_metadata", _mark_diffusion)
assert LlamaCppBackend._gguf_path_is_diffusion("/cache/model.gguf", "owner/model") is True
src = inspect.getsource(llama_cpp_module.LlamaCppBackend.load_model)
preflight = src.index("_preflight_model_path = self._download_gguf(")
teardown = src.index("# ── Phase 1: kill old process")
assert preflight < teardown
assert "model_path = _preflight_model_path or self._download_gguf(" in src
def test_remote_vulkan_diffusion_rejection_keeps_active_server(monkeypatch):
backend = LlamaCppBackend()
killed = []
monkeypatch.setattr(backend, "_find_llama_server_binary", lambda **_kwargs: "/bin/llama")
monkeypatch.setattr(backend, "_is_vulkan_backend", lambda _binary = None: True)
monkeypatch.setattr(backend, "_get_gpu_memory", lambda _binary = None: [(0, 1024, 2048)])
monkeypatch.setattr(
backend,
"_download_gguf",
lambda **_kwargs: "/cache/diffusion.gguf",
)
monkeypatch.setattr(backend, "_gguf_path_is_diffusion", lambda *_args: True)
monkeypatch.setattr(backend, "_kill_process", lambda: killed.append(True))
monkeypatch.setattr(
llama_cpp_module,
"_resolve_repo_id_casing",
lambda repo: repo,
)
monkeypatch.setattr(
llama_cpp_module,
"_hf_offline_if_dns_dead",
lambda: __import__("contextlib").nullcontext(),
)
with pytest.raises(ValueError, match = "DiffusionGemma"):
backend.load_model(
hf_repo = "owner/model",
hf_variant = "Q4_K_M",
model_identifier = "owner/model",
gpu_ids = [0],
)
assert killed == []
def test_start_diffusion_server_resets_tensor_parallel():
# A prior tensor-parallel chat load leaves self._tensor_parallel True (load_model
# phase 1 only kills the process, it skips the unload reset). Diffusion is never
@ -656,18 +744,16 @@ def test_start_diffusion_server_resets_tensor_parallel():
# diffusion re-Apply reloads against stale tensor-parallel state.
src = inspect.getsource(llama_cpp_module.LlamaCppBackend._start_diffusion_server)
assert "self._tensor_parallel = False" in src
assert "self._requested_gpu_ids = list(self._gpu_ids) if self._gpu_ids else None" in src
def test_route_matches_loaded_settings_collapses_diffusion_gpu_ids():
# The route-level reload dedupe mirrors the backend: for a loaded diffusion
# model it compares the request against the single recorded device, not the
# full requested list, or a same-device multi-GPU pick reloads needlessly.
def test_route_matches_loaded_settings_uses_shared_gpu_pin_matcher():
# Route-level and backend race dedupe must share one normalization path so
# raw, effective, and diffusion pins cannot drift apart.
route_src = (Path(_BACKEND_DIR) / "routes" / "inference.py").read_text(encoding = "utf-8")
match_impl = route_src[route_src.index("def _request_matches_loaded_settings") :]
guard = match_impl.index("if llama_backend.is_diffusion:")
collapse = match_impl.index("[sorted(request.gpu_ids)[0]] if request.gpu_ids else None")
compare = match_impl.index("if _req_gpu_ids != llama_backend.gpu_ids:")
assert guard < collapse < compare
assert "if not llama_backend.matches_gpu_ids(request.gpu_ids):" in match_impl
assert "llama_backend._record_matching_gpu_request(request.gpu_ids)" in match_impl
# ── Manual tensor split: child enumeration pinned to the picker's order ──────

View file

@ -130,6 +130,26 @@ class TestResolveRequestedGpuIds(_GpuCacheResetMixin, unittest.TestCase):
):
self.assertEqual(resolve_requested_gpu_ids([]), [1, 3])
def test_vulkan_ordinals_bypass_cuda_parent_visible_validation(self):
# Vulkan build on a CPU-only torch host: no CUDA parent-visible set and a
# zero physical count, yet a valid Vulkan ordinal must not be rejected as
# a CUDA physical id (issue #7239).
with (
patch.dict(os.environ, {}, clear = True),
patch("utils.hardware.hardware.get_physical_gpu_count", return_value = 0),
):
# As a CUDA physical id, [0] is outside the empty parent-visible set.
with self.assertRaises(ValueError):
resolve_requested_gpu_ids([0])
# As Vulkan ordinals, [0] and [0, 1] pass through unchanged.
self.assertEqual(resolve_requested_gpu_ids([0], is_vulkan = True), [0])
self.assertEqual(resolve_requested_gpu_ids([0, 1], is_vulkan = True), [0, 1])
# Malformed ordinals are still rejected.
with self.assertRaisesRegex(ValueError, "duplicate GPU IDs"):
resolve_requested_gpu_ids([0, 0], is_vulkan = True)
with self.assertRaisesRegex(ValueError, "non-negative"):
resolve_requested_gpu_ids([-1], is_vulkan = True)
def test_apply_gpu_ids_only_updates_cuda_visible_devices(self):
with patch.dict(
os.environ,
@ -853,6 +873,171 @@ class TestRouteErrors(unittest.TestCase):
self.assertIn("only supported on CUDA devices", str(exc_info.exception))
def test_inference_route_resolves_gguf_gpu_ids(self):
# GGUF gpu_ids are now supported: /load routes them through the same
# resolution as non-GGUF loads (rejecting only genuinely invalid ids with
# the resolver's actionable message) rather than a blanket "not supported"
# reject, so /validate can stay consistent with /load (#7239).
import utils.hardware.hardware as hardware_mod
inference_route = _load_route_module(
"inference_route_module_for_gguf_gpu_ids_test",
"routes/inference.py",
)
request = LoadRequest(model_path = "unsloth/test.gguf", gpu_ids = [0, 1])
model_config = SimpleNamespace(
is_gguf = True,
is_lora = False,
gguf_hf_repo = None,
gguf_file = "/tmp/test.gguf",
gguf_mmproj_file = None,
gguf_variant = None,
identifier = "unsloth/test.gguf",
display_name = "unsloth/test.gguf",
is_vision = False,
is_audio = False,
audio_type = None,
has_audio_input = False,
)
def _fake_resolve(ids, is_vulkan = False):
raise ValueError("SENTINEL requested GPUs are outside the parent-visible set")
with (
patch.object(
inference_route,
"ModelConfig",
SimpleNamespace(from_identifier = lambda **_kwargs: model_config),
),
# Patch both the package re-export and the defining module so the stub
# fires no matter which import path the route uses.
patch("utils.hardware.resolve_requested_gpu_ids", _fake_resolve),
patch.object(hardware_mod, "resolve_requested_gpu_ids", _fake_resolve),
patch.object(
inference_route,
"_guard_chat_load_against_training",
return_value = None,
),
patch.object(inference_route.asyncio, "to_thread", new = _inline_to_thread),
patch.object(inference_route, "_hf_offline_if_dns_dead", nullcontext),
):
with self.assertRaises(HTTPException) as exc_info:
asyncio.run(
inference_route._load_model_impl(
request,
SimpleNamespace(
app = SimpleNamespace(
state = SimpleNamespace(llama_parallel_slots = 1),
),
),
current_subject = "test-user",
)
)
# The selection was routed through resolution (not the old blanket reject).
self.assertEqual(exc_info.exception.status_code, 400)
self.assertIn("SENTINEL", exc_info.exception.detail)
self.assertNotIn("not supported for GGUF", exc_info.exception.detail)
def test_load_rejects_unavailable_vulkan_ordinal_before_training_guard(self):
inference_route = _load_route_module(
"inference_route_module_for_vulkan_preflight_test",
"routes/inference.py",
)
request = LoadRequest(model_path = "unsloth/test.gguf", gpu_ids = [99])
model_config = SimpleNamespace(
is_gguf = True,
is_lora = False,
gguf_hf_repo = None,
gguf_file = "/tmp/test.gguf",
gguf_mmproj_file = None,
gguf_variant = None,
identifier = "unsloth/test.gguf",
display_name = "unsloth/test.gguf",
is_vision = False,
is_audio = False,
audio_type = None,
has_audio_input = False,
)
with (
patch.object(
inference_route,
"ModelConfig",
SimpleNamespace(from_identifier = lambda **_kwargs: model_config),
),
patch("utils.hardware.get_device", return_value = DeviceType.CUDA),
patch.object(inference_route, "_classify_diffusion_gguf", return_value = None),
patch.object(
inference_route.LlamaCppBackend,
"_is_vulkan_backend",
return_value = True,
),
patch.object(
inference_route.LlamaCppBackend,
"_find_llama_server_binary",
return_value = "/tmp/llama-server",
),
patch.object(
inference_route.LlamaCppBackend,
"_get_gpu_memory",
return_value = [(0, 8 * 1024**3, 16 * 1024**3)],
),
patch.object(
inference_route,
"_guard_chat_load_against_training",
return_value = None,
) as training_guard,
patch.object(inference_route.asyncio, "to_thread", new = _inline_to_thread),
patch.object(inference_route, "_hf_offline_if_dns_dead", nullcontext),
):
with self.assertRaises(HTTPException) as exc_info:
asyncio.run(
inference_route._load_model_impl(
request,
SimpleNamespace(
app = SimpleNamespace(
state = SimpleNamespace(llama_parallel_slots = 1),
),
),
current_subject = "test-user",
)
)
self.assertEqual(exc_info.exception.status_code, 400)
self.assertIn("Vulkan GPU ordinal(s) [99]", exc_info.exception.detail)
training_guard.assert_not_called()
def test_vulkan_ordinals_are_allowed_on_xpu_hosts(self):
import utils.hardware.hardware as hardware_mod
inference_route = _load_route_module(
"inference_route_module_for_xpu_vulkan_test",
"routes/inference.py",
)
config = SimpleNamespace(is_gguf = True)
with (
patch("utils.hardware.get_device", return_value = DeviceType.XPU),
patch.object(
inference_route.LlamaCppBackend,
"_is_vulkan_backend",
return_value = True,
),
patch.object(inference_route, "_classify_diffusion_gguf", return_value = False),
patch.object(hardware_mod, "resolve_requested_gpu_ids", return_value = [0, 1]),
patch.object(
inference_route.LlamaCppBackend,
"_find_llama_server_binary",
return_value = None,
),
):
resolved = asyncio.run(
inference_route._resolve_gguf_gpu_ids_for_request(config, [1, 0])
)
self.assertEqual(resolved, [0, 1])
def test_inference_route_validates_gpu_ids_for_gguf(self):
# gpu_ids is now SUPPORTED for GGUF (the GPU picker), but still
# validated: a rejected pick surfaces as a clean 400, not the old
@ -861,7 +1046,7 @@ class TestRouteErrors(unittest.TestCase):
import utils.hardware.hardware as hardware_mod
inference_route = _load_route_module(
"inference_route_module_for_gguf_gpu_ids_test",
"inference_route_module_for_gguf_gpu_ids_test2",
"routes/inference.py",
)
request = LoadRequest(model_path = "unsloth/test.gguf", gpu_ids = [0, 1])
@ -886,6 +1071,17 @@ class TestRouteErrors(unittest.TestCase):
"ModelConfig",
SimpleNamespace(from_identifier = lambda **_kwargs: model_config),
),
# Patch both the package re-export and the defining module so the stub
# fires no matter which import path the route uses.
patch(
"utils.hardware.resolve_requested_gpu_ids",
side_effect = ValueError("Invalid gpu_ids [0, 1]: rejected by test"),
),
patch.object(
hardware_mod,
"resolve_requested_gpu_ids",
side_effect = ValueError("Invalid gpu_ids [0, 1]: rejected by test"),
),
patch.object(
inference_route,
"_guard_chat_load_against_training",
@ -893,11 +1089,6 @@ class TestRouteErrors(unittest.TestCase):
),
patch.object(inference_route.asyncio, "to_thread", new = _inline_to_thread),
patch.object(inference_route, "_hf_offline_if_dns_dead", nullcontext),
patch.object(
hardware_mod,
"resolve_requested_gpu_ids",
side_effect = ValueError("Invalid gpu_ids [0, 1]: rejected by test"),
),
):
with self.assertRaises(HTTPException) as exc_info:
asyncio.run(

View file

@ -1639,17 +1639,34 @@ def get_parent_visible_gpu_ids() -> list[int]:
return list(parent_visible_ids) if parent_visible_ids is not None else []
def resolve_requested_gpu_ids(gpu_ids: Optional[list[int]]) -> list[int]:
def resolve_requested_gpu_ids(
gpu_ids: Optional[list[int]], *, is_vulkan: bool = False
) -> list[int]:
parent_visible_spec = _get_parent_visible_gpu_spec()
parent_visible_ids = get_parent_visible_gpu_ids()
physical_gpu_count = get_physical_gpu_count()
if gpu_ids is None:
return parent_visible_ids
return [] if is_vulkan else parent_visible_ids
requested_ids = list(gpu_ids)
if len(requested_ids) == 0:
return parent_visible_ids
return [] if is_vulkan else parent_visible_ids
if is_vulkan:
# A Vulkan build selects by ggml Vulkan ordinal (--device VulkanN), a separate
# index space from CUDA/ROCm ids that may be empty under CPU-only torch. The
# CUDA parent-visible / physical-count checks below do not apply; only reject
# malformed ordinals (issue #7239).
if len(set(requested_ids)) != len(requested_ids):
raise ValueError(f"Invalid gpu_ids {requested_ids}: duplicate GPU IDs are not allowed.")
negative_ids = [gpu_id for gpu_id in requested_ids if gpu_id < 0]
if negative_ids:
raise ValueError(
f"Invalid gpu_ids {requested_ids}: GPU IDs must be non-negative. "
f"Rejected IDs: {negative_ids}."
)
return requested_ids
if not parent_visible_spec["supports_explicit_gpu_ids"]:
raise ValueError(
@ -2193,12 +2210,13 @@ def auto_select_gpu_ids(
metadata["selection_mode"] = "auto"
metadata["selected_gpu_ids"] = selected
logger.debug(
"Selected GPUs automatically",
model_name = model_name,
selected_gpu_ids = selected,
usable_gb = metadata["usable_gb"],
required_gb = metadata.get("required_gb"),
multi_gpu_overhead = multi_gpu_overhead,
"Selected GPUs automatically: model=%s selected=%s usable_gb=%s "
"required_gb=%s multi_gpu_overhead=%s",
model_name,
selected,
metadata["usable_gb"],
metadata.get("required_gb"),
multi_gpu_overhead,
)
return selected, metadata
@ -2214,12 +2232,13 @@ def auto_select_gpu_ids(
metadata["usable_gb"] = round(fallback_usable, 3)
metadata["selected_gpu_ids"] = fallback_all
logger.warning(
"Falling back to all visible GPUs -- model may not fit",
model_name = model_name,
selected_gpu_ids = fallback_all,
usable_gb = metadata["usable_gb"],
required_gb = metadata.get("required_gb"),
multi_gpu_overhead = multi_gpu_overhead,
"Falling back to all visible GPUs; model may not fit: model=%s "
"selected=%s usable_gb=%s required_gb=%s multi_gpu_overhead=%s",
model_name,
fallback_all,
metadata["usable_gb"],
metadata.get("required_gb"),
multi_gpu_overhead,
)
return fallback_all, metadata

View file

@ -222,7 +222,9 @@ export function applyActiveModelStatusToStore(
incomingGpuMode === "manual" ? (status.n_cpu_moe ?? null) : null;
const incomingSplit =
incomingGpuMode === "manual" ? (status.tensor_split ?? null) : null;
const incomingGpuIds = status.is_gguf ? (status.gpu_ids ?? null) : null;
const incomingGpuIds = status.is_gguf
? (status.requested_gpu_ids ?? status.gpu_ids ?? null)
: null;
const gpuStatusChanged =
prevState.loadedGpuMemoryMode !== incomingGpuMode ||
prevState.loadedGpuLayers !== incomingGpuLayers ||

View file

@ -605,6 +605,7 @@ export function loadedGpuMemoryFields(resp: {
n_layers?: number | null;
n_moe_layers?: number;
gpu_ids?: number[] | null;
requested_gpu_ids?: number[] | null;
}) {
// GPU-memory state is meaningful only for a GGUF chat load. A non-GGUF response
// still carries gpu_memory_mode (its default "auto" is serialized), so gate on
@ -631,7 +632,9 @@ export function loadedGpuMemoryFields(resp: {
};
}
const mode = resp.gpu_memory_mode ?? "auto";
const gpuIds = resp.gpu_ids ?? null;
// Keep the user's placement pool editable across status/load hydration.
// gpu_ids remains the effective fitted subset for diagnostics.
const gpuIds = resp.requested_gpu_ids ?? resp.gpu_ids ?? null;
// Layer/MoE/split knobs apply (and are reported) only in manual mode; in auto
// the server ignores them, so don't seed the loaded baseline or the editable
// knobs with values it never applied. In manual, the server reports gpu_layers
@ -669,7 +672,7 @@ export function loadedGpuMemoryFields(resp: {
ggufLayerCount: resp.n_layers ?? null,
// MoE expert-layer count: the n_cpu_moe slider max, and 0 hides the slider.
moeLayerCount: resp.n_moe_layers ?? null,
// The picker reflects what loaded (the request sent the user's pick).
// The picker reflects the requested placement pool, not a fitted subset.
selectedGpuIds: gpuIds,
loadedGpuIds: gpuIds,
...manualKnobs,

View file

@ -188,7 +188,10 @@ export interface LoadModelResponse {
n_layers?: number | null;
/** Model's MoE expert-layer count (the n_cpu_moe ceiling); 0 if not MoE. */
n_moe_layers?: number;
/** Effective GPU placement after fit-time narrowing. */
gpu_ids?: number[] | null;
/** User-requested GPU placement pool before fit-time narrowing. */
requested_gpu_ids?: number[] | null;
}
export interface UnloadModelRequest {
@ -240,7 +243,10 @@ export interface InferenceStatusResponse {
/** n_ctx the active GGUF load was invoked with (0 = Auto); re-seeds a
* Manual + Auto-layers context pin on hydration. Null for non-GGUF. */
requested_context_length?: number | null;
/** Effective GPU placement after fit-time narrowing. */
gpu_ids?: number[] | null;
/** User-requested GPU placement pool before fit-time narrowing. */
requested_gpu_ids?: number[] | null;
n_layers?: number | null;
/** Model's MoE expert-layer count (the n_cpu_moe ceiling); 0 if not MoE. */
n_moe_layers?: number;

View file

@ -151,6 +151,19 @@ def test_active_model_config_round_trips_gpu_fields():
assert "export function gpuFieldsSignature" in shared
def test_gpu_picker_round_trips_requested_pool_not_fitted_subset():
"""A GGUF fit may narrow [0, 1] to [0], but load/status hydration must keep
[0, 1] as the editable pool so a later reload can grow back onto GPU 1."""
types = _read("features/chat/types/api.ts")
assert types.count("requested_gpu_ids?: number[] | null") >= 2
store = _read("features/chat/stores/chat-runtime-store.ts")
assert "resp.requested_gpu_ids ?? resp.gpu_ids ?? null" in store
status = _read("features/chat/lib/apply-inference-status-to-store.ts")
assert "status.requested_gpu_ids ?? status.gpu_ids ?? null" in status
def test_compare_load_uses_each_models_gpu_config():
src = _read("features/chat/shared-composer.tsx")
assert "ownConfig.gpuMemoryMode ?? compareLoadKnobs.gpuMemoryMode" in src