Studio: make explicit offloaded-layer count authoritative in GPU classifier

Reviewers found a real CPU-only log shape that passed: an 'offloading 0
repeating layers to GPU' planning line, or a GPU KV/compute buffer, could read
as GPU before the definitive 'offloaded 0/33' was seen. Check the explicit
counted offload first (any N>0 wins, all zero is CPU-only), restrict the
buffer-size signal to GPU model buffers (KV/compute on GPU with weights on CPU
is still CPU inference), and add HIP/MUSA/CANN to the model-buffer markers so an
older log naming those backends is not misread as CPU. Same in both classifiers.
This commit is contained in:
danielhanchen 2026-06-01 16:14:46 +00:00
commit 1546f0328e
4 changed files with 144 additions and 61 deletions

View file

@ -65,7 +65,18 @@ logger = get_logger(__name__)
# #5106 / #5830). Robust across llama.cpp log formats (the "model buffer size"
# lines were dropped in recent builds): buffer-size markers, "offloaded N/M
# layers to GPU", and the "device_info:" enumeration.
_GPU_BUFFER_MARKERS = ("CUDA", "ROCm", "Metal", "Vulkan", "OpenCL", "SYCL")
_GPU_BUFFER_MARKERS = (
"CUDA",
"ROCm",
"ROCM",
"HIP",
"Metal",
"Vulkan",
"OpenCL",
"SYCL",
"MUSA",
"CANN",
)
_DEVICE_ROW_RE = re.compile(
r"-\s*(?P<dev>(?:CUDA|ROCm|ROCM|HIP|Metal|Vulkan|SYCL|OpenCL|MUSA|CANN|CPU)\w*)\s*:",
re.IGNORECASE,
@ -84,40 +95,49 @@ _GPU_DEVICE_PREFIXES = (
_OFFLOADED_LAYERS_RE = re.compile(
r"offloaded\s+(\d+)\s*/\s*(\d+)\s+layers?\s+to\s+gpu", re.IGNORECASE
)
_OFFLOADING_COUNT_RE = re.compile(
r"offloading\s+(\d+)\s+(?:repeating\s+|non-repeating\s+)?layers?\s+to\s+gpu",
re.IGNORECASE,
)
def classify_gpu_offload_lines(lines: list[str]) -> Optional[bool]:
"""True if the model landed on a GPU, False if it stayed on CPU despite GPU
intent, None when the log has no usable signal. Priority: buffer-size lines,
then offloaded-layers count, then device_info enumeration."""
# Exclude host-pinned buffers ("CUDA_Host" ...): CPU RAM the GPU backend
# pinned, not device memory, so they must not read as GPU offload.
saw_buffer_line = False
intent, None when the log has no usable signal. Priority: explicit
offloaded-layer counts (authoritative), then GPU model-buffer lines, then
device_info enumeration."""
# Signal 1: explicit offloaded counts win over everything (a KV/compute
# buffer can be on the GPU while 0 model layers are offloaded). Any counted
# line N>0 => True; all 0 => False (scan all; a draft model can log 0/k
# before the main model's 33/33). Uncounted "offloading output layer to GPU"
# is only a weak positive when no counted line exists.
saw_zero_count = False
saw_uncounted_offloading = False
for line in lines:
if "buffer size" not in line:
match = _OFFLOADED_LAYERS_RE.search(line) or _OFFLOADING_COUNT_RE.search(line)
if match:
if int(match.group(1)) > 0:
return True
saw_zero_count = True
continue
low = line.lower()
if "offloading" in low and "to gpu" in low:
saw_uncounted_offloading = True
if saw_zero_count:
return False
if saw_uncounted_offloading:
return True
# Signal 2: GPU marker on a *model* buffer (exclude host-pinned _Host).
saw_model_buffer = False
for line in lines:
if "model buffer size" not in line:
continue
saw_model_buffer = True
if "_Host" not in line and any(
marker in line for marker in _GPU_BUFFER_MARKERS
):
return True
if "model buffer size" in line:
saw_buffer_line = True
# Accept if any "offloaded N/M" has N>0 (a draft model can log 0/k before
# the main model's 33/33); CPU-only only when every offloaded line is zero.
saw_offloaded = False
for line in lines:
match = _OFFLOADED_LAYERS_RE.search(line)
if match:
saw_offloaded = True
if int(match.group(1)) > 0:
return True
continue
low = line.lower()
if "offloading" in low and "to gpu" in low:
return True
if saw_offloaded:
return False
after_device_info = False
saw_device_row = False
@ -135,7 +155,7 @@ def classify_gpu_offload_lines(lines: list[str]) -> Optional[bool]:
if any(dev.startswith(prefix) for prefix in _GPU_DEVICE_PREFIXES):
return True
if saw_buffer_line or saw_device_row:
if saw_model_buffer or saw_device_row:
return False
return None

View file

@ -640,3 +640,19 @@ class TestClassifyGpuOffload:
]
)
assert inst._classify_gpu_offload(True, [(0, 22805)]) is True
def test_offloading_zero_repeating_does_not_mask_zero_total(self):
inst = self._backend(
[
"llm_load_tensors: offloading 0 repeating layers to GPU",
"llm_load_tensors: offloaded 0/33 layers to GPU",
"llm_load_tensors: CPU model buffer size = 7338.64 MiB",
]
)
assert inst._classify_gpu_offload(True, [(0, 22805)]) is False
def test_hip_model_buffer_is_gpu(self):
inst = self._backend(
["load_tensors: HIP0 model buffer size = 21000.0 MiB"]
)
assert inst._classify_gpu_offload(True, [(0, 22805)]) is True

View file

@ -5604,7 +5604,18 @@ def validate_quantize(
# llama.cpp). CUDA0 / ROCm0 / Metal / Vulkan0 / OpenCL0 / SYCL0 / HIP0 / MUSA0
# / CANN0 are GPU; CPU / CPU_Mapped are not. Kept broad so a single helper
# covers every backend Studio ships.
_GPU_MODEL_BUFFER_MARKERS = ("CUDA", "ROCm", "Metal", "Vulkan", "OpenCL", "SYCL")
_GPU_MODEL_BUFFER_MARKERS = (
"CUDA",
"ROCm",
"ROCM",
"HIP",
"Metal",
"Vulkan",
"OpenCL",
"SYCL",
"MUSA",
"CANN",
)
# A device_info row looks like "<ts> I - CUDA0 : NVIDIA B200 (...)" or
# "<ts> I - CPU : ...". Matched on the "- <backend> :" shape (the leading
@ -5630,10 +5641,18 @@ _GPU_DEVICE_PREFIXES = (
"cann",
)
# "load_tensors: offloaded 33/33 layers to GPU" (or "offloading ... to GPU").
# "load_tensors: offloaded 33/33 layers to GPU".
_OFFLOADED_LAYERS_RE = re.compile(
r"offloaded\s+(\d+)\s*/\s*(\d+)\s+layers?\s+to\s+gpu", re.IGNORECASE
)
# "offloading 0 repeating layers to GPU" / "offloading 1 non-repeating ...".
# A counted form, so a zero here is a definite CPU-only signal (it usually
# precedes "offloaded 0/N"); kept separate from the uncounted
# "offloading output layer to GPU" phrasing, which carries no number.
_OFFLOADING_COUNT_RE = re.compile(
r"offloading\s+(\d+)\s+(?:repeating\s+|non-repeating\s+)?layers?\s+to\s+gpu",
re.IGNORECASE,
)
# install_kind values that ship a real GPU backend and must offload to the GPU.
# A binary launched with --n-gpu-layers 1 under one of these is held to the
@ -5649,9 +5668,12 @@ def server_log_shows_gpu_offload(log_text: str) -> bool | None:
Robust across llama.cpp log formats (the "model buffer size" lines were
dropped in recent builds), in priority order:
1. ``<backend> model buffer size = N`` lines -- GPU marker => True
(older llama.cpp).
2. ``offloaded N/M layers to GPU`` -- N>0 => True, 0/M => False.
1. ``offloaded N/M layers to GPU`` / ``offloading N ... layers to GPU`` --
the explicit layer count is authoritative: N>0 => True, all 0 => False.
It must win over auxiliary GPU buffers (a KV/compute buffer can sit on
the GPU while 0 model layers are offloaded -- still CPU inference).
2. ``<backend> model buffer size = N`` lines -- a GPU marker on a *model*
buffer (not host-pinned ``CUDA_Host``) => True (older llama.cpp).
3. ``device_info:`` enumeration -- a GPU device row (``- CUDA0 :`` etc.)
=> True; only a ``- CPU :`` row => False. This is the version-stable
signal and matches how the CPU-only-binary bug is diagnosed in the
@ -5663,41 +5685,43 @@ def server_log_shows_gpu_offload(log_text: str) -> bool | None:
"""
lines = log_text.splitlines()
# Signal 1: per-backend buffer-size lines. A GPU marker on ANY "buffer
# size" line (model / KV / compute) means the GPU holds part of the model
# -- accept. Exclude host-pinned buffers ("CUDA_Host" / "ROCm_Host" ...):
# those are CPU RAM the GPU backend pinned, not device memory, so a binary
# that pins host memory but offloads no weights must not read as GPU. Only
# the model-buffer location decides CPU-only (KV/compute CPU buffers exist
# even on GPU runs), so the False determination keys on "model buffer size".
saw_buffer_line = False
# Signal 1: explicit offloaded-layers counts (authoritative). Any counted
# line with N>0 => True; if every counted line is 0 => False (a draft model
# can log "offloaded 0/k" before the main model's "offloaded 33/33", so scan
# all before deciding). The uncounted "offloading output layer to GPU"
# phrasing carries no number, so it is only a weak positive used when no
# counted line exists at all.
saw_zero_count = False
saw_uncounted_offloading = False
for line in lines:
if "buffer size" not in line:
match = _OFFLOADED_LAYERS_RE.search(line) or _OFFLOADING_COUNT_RE.search(line)
if match:
if int(match.group(1)) > 0:
return True
saw_zero_count = True
continue
low = line.lower()
if "offloading" in low and "to gpu" in low:
saw_uncounted_offloading = True
if saw_zero_count:
return False
if saw_uncounted_offloading:
return True
# Signal 2: per-backend model-buffer lines. A GPU marker on a *model* buffer
# means model weights live on the GPU. Exclude host-pinned buffers
# ("CUDA_Host" / "ROCm_Host" ...): those are CPU RAM the GPU backend pinned,
# not device memory. KV/compute buffers are ignored here -- they can be on
# the GPU even when all weights are on CPU.
saw_model_buffer = False
for line in lines:
if "model buffer size" not in line:
continue
saw_model_buffer = True
if "_Host" not in line and any(
marker in line for marker in _GPU_MODEL_BUFFER_MARKERS
):
return True
if "model buffer size" in line:
saw_buffer_line = True
# Signal 2: explicit offloaded-layers count. Scan every "offloaded N/M"
# line and accept if any has N>0 (a draft/speculative model can log
# "offloaded 0/k" before the main model's "offloaded 33/33"); only when all
# offloaded lines are zero is it CPU-only.
saw_offloaded = False
for line in lines:
match = _OFFLOADED_LAYERS_RE.search(line)
if match:
saw_offloaded = True
if int(match.group(1)) > 0:
return True
continue
low = line.lower()
if "offloading" in low and "to gpu" in low:
return True
if saw_offloaded:
return False
# Signal 3: device_info enumeration. Only trust device rows once the
# "device_info:" header has appeared, so the compiled-backend system_info
@ -5718,7 +5742,7 @@ def server_log_shows_gpu_offload(log_text: str) -> bool | None:
if any(dev.startswith(prefix) for prefix in _GPU_DEVICE_PREFIXES):
return True
if saw_buffer_line or saw_device_row:
if saw_model_buffer or saw_device_row:
# We had a concrete signal, but every buffer/device was CPU.
return False
return None

View file

@ -216,17 +216,40 @@ def test_signal2_offloaded_zero_beats_device_info():
assert server_log_shows_gpu_offload(log) is False
def test_gpu_buffer_size_without_model_word():
# A GPU-marked buffer line that omits the word "model" (e.g. KV/compute or
# a future format) must still count as GPU offload, even when a CPU
# "model buffer size" line is present (broadened signal 1).
def test_kv_buffer_on_gpu_with_cpu_model_is_not_offload():
# A GPU KV/compute buffer while the model weights sit on CPU is still CPU
# inference; only a GPU *model* buffer counts as offload.
log = (
"load_tensors: CUDA0 KV buffer size = 100.0 MiB\n"
"load_tensors: CPU_Mapped model buffer size = 0.6 MiB\n"
)
assert server_log_shows_gpu_offload(log) is False
def test_offloading_zero_repeating_does_not_mask_zero_total():
# Real CPU-only shape: a planning line says "offloading 0 repeating layers"
# before the definitive "offloaded 0/33". The count must decide -> False.
log = (
"llm_load_tensors: offloading 0 repeating layers to GPU\n"
"llm_load_tensors: offloaded 0/33 layers to GPU\n"
"llm_load_tensors: CPU model buffer size = 7338.64 MiB\n"
)
assert server_log_shows_gpu_offload(log) is False
def test_uncounted_offloading_output_layer_is_gpu():
# The uncounted "offloading output layer to GPU" phrasing (no number, no
# later zero count) is a weak positive.
log = "llm_load_tensors: offloading output layer to GPU\n"
assert server_log_shows_gpu_offload(log) is True
def test_hip_musa_cann_model_buffer_is_gpu():
for marker in ("HIP0", "MUSA0", "CANN0"):
log = f"load_tensors: {marker} model buffer size = 21000.0 MiB\n"
assert server_log_shows_gpu_offload(log) is True, marker
def test_device_row_case_insensitive():
log = "device_info:\n - cuda0 : some gpu (free)\n - CPU : x (free)\n"
assert server_log_shows_gpu_offload(log) is True